extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,279 @@
1
+ /**
2
+ * @fileoverview Main orchestrator for the SEEKTOPIC keyphrase extraction pipeline.
3
+ * Coordinates cleaning, segmentation, LLM extraction, and fallback n-gram ranking.
4
+ */
5
+ import type {
6
+ NgramMap,
7
+ KeyphraseEntry,
8
+ SentenceEntry,
9
+ SEEKTOPICOptions,
10
+ SEEKTOPICResult,
11
+ } from "./types";
12
+ import { splitTextToSentences } from "../tokenize/text-to-sentences";
13
+ import { convertTextToTokens } from "../tokenize/text-to-topic-tokens";
14
+ import { extractNounEdgeGrams } from "./ngrams";
15
+ import { foldSubphrases } from "./fold-keyphrases";
16
+ import { weightKeyphrasesBySpecificity } from "./weight-keyphrases";
17
+ import { rankSentencesCentralToKeyphrase } from "./rank-sentences-keyphrases";
18
+ import { weighRelevanceConceptVectorMultiple } from "./vector-search";
19
+
20
+ /**
21
+ * ### SEEKTOPIC \u2014 Keyphrase & Sentence Extraction
22
+ *
23
+ * Pulls the most important phrases and sentences out of any document.
24
+ * Given raw text, it returns a ranked list of key concepts (e.g. "neural
25
+ * network", "climate change") and, optionally, the sentences that best
26
+ * summarise the document around those concepts.
27
+ *
28
+ * <img src="https://i.imgur.com/gZ4kI1V.png" width="360px" />
29
+ *
30
+ * **How it works \u2014 8-step pipeline:**
31
+ *
32
+ * 1. **Clean** \u2014 strip HTML tags and entities.
33
+ * 2. **Split** \u2014 break text into sentences, respecting abbreviations and URLs.
34
+ * 3. **Topic Extraction** *(LLM Path)* \u2014 sends the first 5000 words to an LLM
35
+ * to extract the most descriptive key topic phrases labeling the document.
36
+ * 4. **Vector Search** *(LLM Path)* \u2014 calculates cosine similarity embeddings
37
+ * for all sentences vs topic phrases, finding the top relevant sentences per topic.
38
+ * 5. **Tokenise & Extract** *(Fallback)* \u2014 if LLM fails, identify noun-anchored
39
+ * n-grams (1-N words).
40
+ * 6. **Score & Fold** *(Fallback)* \u2014 weight phrases and merge subsumed phrases.
41
+ * 7. **Re-weight** *(Fallback)* \u2014 boost Wikipedia entities and rare domain terms.
42
+ * 8. **TextRank** *(Fallback)* \u2014 build sentence similarity graph and run random walk.
43
+ *
44
+ * <video src="https://github.com/user-attachments/assets/73348d63-7671-4e20-8df9-29a13d5b0768"
45
+ * width="550px" controls />
46
+ *
47
+ * @param docText - Plain text or HTML document to analyse.
48
+ * @param options - Optional tuning parameters (word limits, thresholds, query bias).
49
+ * @returns
50
+ * - `optionSkipRanking: true` (default) \u2192 `KeyphraseEntry[]` sorted by weight.
51
+ * - `optionSkipRanking: false` \u2192 `SEEKTOPICResult` with `topSentences`,
52
+ * `keyphrases`, and the full `sentences` array.
53
+ *
54
+ * @example
55
+ * // Fast: just extract keyphrases
56
+ * const keyphrases = extractSEEKTOPIC(articleText, { phrasesModel });
57
+ * // \u2192 [{ keyphrase: "machine learning", weight: 84 }, ...]
58
+ *
59
+ * @example
60
+ * // Full: keyphrases + summary sentences, biased toward a search query
61
+ * const { topSentences, keyphrases } = extractSEEKTOPIC(articleText, {
62
+ * phrasesModel,
63
+ * optionSkipRanking: false,
64
+ * heavyWeightQuery: "transformer attention",
65
+ * limitTopSentences: 5,
66
+ * }) as SEEKTOPICResult;
67
+ *
68
+ * @author [ai-research-agent (2024)](https://airesearch.js.org)
69
+ * @category Topics
70
+ */
71
+ export async function extractSEEKTOPIC(
72
+ docText: string,
73
+ options: SEEKTOPICOptions = {},
74
+ ): Promise<KeyphraseEntry[] | SEEKTOPICResult> {
75
+ if (typeof docText !== "string") throw new Error("docText must be a string");
76
+
77
+ const {
78
+ phrasesModel,
79
+ maxWords = 2,
80
+ minWords = 1,
81
+ minWordLength = 3,
82
+ topKeyphrasesPercent = 0.5,
83
+ limitTopSentences = 5,
84
+ limitTopKeyphrases = 10,
85
+ minKeyPhraseLength = 5,
86
+ heavyWeightQuery = "",
87
+ removeHTML = true,
88
+ optionSkipRanking = true,
89
+ getEnv = () => "",
90
+ } = options;
91
+
92
+ // \u2500\u2500 1. Normalize HTML \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
93
+ let text = docText
94
+ .replace(/</g, " <")
95
+ .replace(/>/g, "> ")
96
+ .replace(/&.{2,5};/g, ""); // strip &quot; &amp; &lt; &gt; &nbsp;
97
+ if (removeHTML) text = text.replace(/<[^>]*>/g, "");
98
+
99
+ // \u2500\u2500 2. Sentence segmentation \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
100
+ const sentencesArray = splitTextToSentences(text);
101
+
102
+ // \u2500\u2500 3. LLM Topic Extraction (Starting Step) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
103
+ const first5kWords = text.split(/\s+/).slice(0, 5000).join(" ");
104
+ const llmPrompt = `Extract the 5-10 most important key topic phrases that best
105
+ label this document. Return exclusively a comma-separated list of the key topic
106
+ phrases. Document text: ${first5kWords}`;
107
+
108
+ let llmTopics: string[] = [];
109
+ try {
110
+ // `generate-language` moved into the chat-agent-toolkit workspace package.
111
+ // Load it lazily (and tolerate absence) since the LLM path is an optional
112
+ // enhancement — the n-gram fallback below covers extraction without it.
113
+ const { writeLanguageResponse } = await import("chat-agent-toolkit");
114
+ const aiResponse = await writeLanguageResponse({
115
+ query: llmPrompt,
116
+ provider: getEnv("LLM_PROVIDER") || "openai",
117
+ apiKey: getEnv("LLM_API_KEY") || getEnv("OPENAI_API_KEY") || "dummy",
118
+ });
119
+ if (aiResponse.content && !aiResponse.error) {
120
+ llmTopics = aiResponse.content
121
+ .split(",")
122
+ .map((s: string) => s.trim().replace(/^['"]|['"]$/g, ""))
123
+ .filter(Boolean);
124
+ }
125
+ } catch (e) {
126
+ console.error("LLM topic extraction failed", e);
127
+ }
128
+
129
+ // \u2500\u2500 4. Use vector-search.ts for top topic sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
130
+ if (llmTopics.length > 0) {
131
+ if (optionSkipRanking) {
132
+ return llmTopics.map((topic: string) => ({
133
+ keyphrase: topic,
134
+ sentences: [],
135
+ words: topic.split(/\s+/).length,
136
+ weight: 100,
137
+ }));
138
+ }
139
+
140
+ const { limitTopSentences = 3 } = options;
141
+ const allRelevant = await weighRelevanceConceptVectorMultiple(
142
+ sentencesArray,
143
+ llmTopics,
144
+ options,
145
+ );
146
+
147
+ const keyphraseObjects: any[] = [];
148
+ const topSentences: any[] = [];
149
+
150
+ for (const topic of llmTopics) {
151
+ // @ts-ignore
152
+ const topicRelevant = allRelevant[topic] || [];
153
+ const topX = topicRelevant.slice(0, limitTopSentences);
154
+
155
+ const topicSentences = topX.map((r: any) => ({
156
+ text: r.content,
157
+ similarity: r.similarity,
158
+ }));
159
+
160
+ keyphraseObjects.push({
161
+ keyphrase: topic,
162
+ topSentences: topicSentences,
163
+ words: topic.split(/\s+/).length,
164
+ weight: 100,
165
+ });
166
+
167
+ for (const r of topX) {
168
+ const sIdx = sentencesArray.indexOf(r.content);
169
+ topSentences.push({
170
+ text: r.content,
171
+ index: sIdx !== -1 ? sIdx : 0,
172
+ keyphrases: [{ keyphrase: topic, weight: r.similarity }],
173
+ weight: r.similarity,
174
+ });
175
+ }
176
+ }
177
+
178
+ // Deduplicate topSentences
179
+ const uniqueSentences = Array.from(
180
+ new Map(topSentences.map((s: any) => [s.text, s])).values(),
181
+ );
182
+
183
+ return {
184
+ topSentences: uniqueSentences.sort(
185
+ (a: any, b: any) => b.weight - a.weight,
186
+ ),
187
+ keyphrases: keyphraseObjects,
188
+ sentences: sentencesArray,
189
+ } as unknown as SEEKTOPICResult;
190
+ }
191
+
192
+ // \u2500\u2500 FALLBACK: Tokenise + extract n-grams \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
193
+ const nGrams: NgramMap = {};
194
+ const sentenceKeysMap: SentenceEntry[] = [];
195
+
196
+ for (let idx = 0; idx < sentencesArray.length; idx++) {
197
+ const sentence = sentencesArray[idx];
198
+ const tokens = convertTextToTokens(sentence, { phrasesModel });
199
+
200
+ sentenceKeysMap.push({
201
+ text: sentence,
202
+ index: idx,
203
+ keyphrases: [],
204
+ weight: 0,
205
+ });
206
+
207
+ for (let i = 0; i < tokens.length; i++) {
208
+ for (let n = minWords; n <= maxWords; n++) {
209
+ extractNounEdgeGrams(n, tokens, i, nGrams, minWordLength, idx);
210
+ }
211
+ }
212
+ }
213
+
214
+ // \u2500\u2500 5. Score raw keyphrases \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
215
+ let keyphraseObjects: KeyphraseEntry[] = [];
216
+ for (let n = minWords; n <= maxWords; n++) {
217
+ if (!nGrams[n]) continue;
218
+ for (const [keyphrase, sentencesArr] of Object.entries(nGrams[n])) {
219
+ keyphraseObjects.push({
220
+ keyphrase,
221
+ sentences: sentencesArr,
222
+ words: n,
223
+ weight: sentencesArr.length * n,
224
+ });
225
+ }
226
+ }
227
+
228
+ // \u2500\u2500 6. Fold subphrases + deduplicate \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
229
+ const folded = foldSubphrases(keyphraseObjects);
230
+ const seen = new Set<string>();
231
+ keyphraseObjects = folded
232
+ .sort((a, b) => b.weight - a.weight)
233
+ .filter((k) => !seen.has(k.keyphrase) && seen.add(k.keyphrase))
234
+ .map((k) => ({
235
+ ...k,
236
+ sentences: Array.isArray(k.sentences)
237
+ ? [...new Set(k.sentences as number[])]
238
+ : k.sentences,
239
+ }));
240
+
241
+ // \u2500\u2500 7. IDF + wiki-entity + heavy-query weighting \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
242
+ keyphraseObjects = weightKeyphrasesBySpecificity(
243
+ keyphraseObjects,
244
+ phrasesModel,
245
+ heavyWeightQuery,
246
+ )
247
+ .filter((k) => k.keyphrase.length > minKeyPhraseLength)
248
+ .sort((a, b) => b.weight - a.weight);
249
+
250
+ // Fast path: return keyphrases without TextRank
251
+ if (optionSkipRanking) return keyphraseObjects;
252
+
253
+ // \u2500\u2500 8. Attach keyphrases to sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
254
+ for (const { keyphrase, sentences, weight } of keyphraseObjects) {
255
+ if (Array.isArray(sentences)) {
256
+ for (const si of sentences) {
257
+ sentenceKeysMap[si as number]?.keyphrases.push({ keyphrase, weight });
258
+ }
259
+ }
260
+ }
261
+
262
+ // \u2500\u2500 9. TextRank \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
263
+ const ranked = rankSentencesCentralToKeyphrase(sentenceKeysMap);
264
+
265
+ const topSentences = (ranked ?? [])
266
+ .sort((a, b) => b.weight - a.weight)
267
+ .slice(0, limitTopSentences)
268
+ .map((s) => ({
269
+ ...s,
270
+ keyphrases: [...new Set(s.keyphrases.map((k) => k.keyphrase))],
271
+ }));
272
+
273
+ const keyphrases = keyphraseObjects.slice(0, limitTopKeyphrases).map((k) => ({
274
+ ...k,
275
+ sentences: Array.isArray(k.sentences) ? k.sentences.join(",") : k.sentences,
276
+ }));
277
+
278
+ return { topSentences, keyphrases, sentences: sentencesArray };
279
+ }
@@ -0,0 +1,92 @@
1
+ /**
2
+ * @fileoverview Type definitions for the SEEKTOPIC engine and ranking utilities.
3
+ */
4
+ import type {
5
+ TopicToken,
6
+ PhrasesModel,
7
+ } from "../tokenize/text-to-topic-tokens";
8
+
9
+ export type { TopicToken, PhrasesModel };
10
+
11
+ /** Map from n-gram size \u2192 { phrase text \u2192 sentence indices } */
12
+ export type NgramMap = Record<number, Record<string, number[]>>;
13
+
14
+ /**
15
+ * A ranked keyphrase with its scoring metadata and the sentence indices it appears in.
16
+ */
17
+ export interface KeyphraseEntry {
18
+ /** The keyphrase text */
19
+ keyphrase: string;
20
+ /** Indices of sentences containing this keyphrase (or comma-separated string form) */
21
+ sentences: number[] | string;
22
+ /** Sentences representing this keyphrase directly with their relevance similarity */
23
+ topSentences?: Array<{ text: string; similarity: number }>;
24
+ /** Number of words in the keyphrase */
25
+ words: number;
26
+ /** Composite ranking weight */
27
+ weight: number;
28
+ /** True if the phrase is a Wikipedia-linked entity */
29
+ wiki?: boolean;
30
+ }
31
+
32
+ /**
33
+ * A sentence node used during graph construction and TextRank scoring.
34
+ */
35
+ export interface SentenceEntry {
36
+ text: string;
37
+ index: number;
38
+ keyphrases: Array<{ keyphrase: string; weight: number }>;
39
+ weight: number;
40
+ }
41
+
42
+ /**
43
+ * A sentence in the final SEEKTOPIC output \u2014 keyphrases are resolved to strings.
44
+ */
45
+ export interface SentenceResult {
46
+ text: string;
47
+ index: number;
48
+ keyphrases: string[];
49
+ weight: number;
50
+ }
51
+
52
+ /** Options for {@link extractSEEKTOPIC} */
53
+ export interface SEEKTOPICOptions {
54
+ /** Trie phrases model for wiki-phrase tokenization */
55
+ phrasesModel?: PhrasesModel;
56
+ /** Maximum words per keyphrase (default 2) */
57
+ maxWords?: number;
58
+ /** Minimum words per keyphrase (default 1) */
59
+ minWords?: number;
60
+ /** Minimum character length of any word in a keyphrase (default 3) */
61
+ minWordLength?: number;
62
+ /** Fraction of all keyphrases to use as the TextRank graph limit (default 0.5) */
63
+ topKeyphrasesPercent?: number;
64
+ /** Max sentences to return in full-ranking mode (default 5) */
65
+ limitTopSentences?: number;
66
+ /** Max keyphrases to return (default 10) */
67
+ limitTopKeyphrases?: number;
68
+ /** Minimum character length of the full keyphrase string (default 5) */
69
+ minKeyPhraseLength?: number;
70
+ /** Query string to bias keyphrase weights toward (default "") */
71
+ heavyWeightQuery?: string;
72
+ /** Strip HTML tags before processing (default true) */
73
+ removeHTML?: boolean;
74
+ /** Return only keyphrases without running TextRank (default true) */
75
+ optionSkipRanking?: boolean;
76
+ getEnv?: (key: string) => string | undefined;
77
+ }
78
+
79
+ /** Full SEEKTOPIC output when `optionSkipRanking` is false */
80
+ export interface SEEKTOPICResult {
81
+ topSentences: SentenceResult[];
82
+ keyphrases: Array<Omit<KeyphraseEntry, "sentences"> & { sentences: string }>;
83
+ sentences: string[];
84
+ }
85
+
86
+ /** Options for {@link rankSentencesCentralToKeyphrase} */
87
+ export interface RankOptions {
88
+ /** Number of random-walk steps (default 1000) */
89
+ iterations?: number;
90
+ /** Steps between forced random resets to avoid cluster traps (default 100) */
91
+ resetInterval?: number;
92
+ }
@@ -0,0 +1,232 @@
1
+ import grab from "../utils/grab";
2
+
3
+ /**
4
+ * Text embeddings convert words or phrases into numerical vectors in a high-dimensional
5
+ * space, where each dimension represents a semantic feature extracted by a model like
6
+ * MiniLM-L6-v2. In this concept space, words with similar meanings have vectors that
7
+ * are close together, allowing for quantitative comparisons of semantic similarity.
8
+ * These vector representations enable powerful applications in natural language processing,
9
+ * including semantic search, text classification, and clustering, by leveraging the
10
+ * geometric properties of the embedding space to capture and analyze the relationships
11
+ * between words and concepts.
12
+ * [Text Embeddings, Classification, and Semantic Search
13
+ * (Youtube)](https://www.youtube.com/watch?v=sNa_uiqSlJo&t=129s)
14
+ *
15
+ * <img src="https://i.imgur.com/wtJqEqX.png" width="350" />
16
+ * @param {string} text - The text to embed.
17
+ * @param {Object} [options]
18
+ * @param {AutoTokenizer} options.pipeline
19
+ * - The pipeline to use for embedding.
20
+ * @param {number} options.precision default=4 - The number of decimal places to round to.
21
+ * @returns {Promise<{embeddingsDict: Object.<string, number[]>, embedding: number[]}>}
22
+ * @category Similarity
23
+ */
24
+ export async function convertTextToEmbedding(
25
+ text: string | string[],
26
+ options: any = {},
27
+ ) {
28
+ var { precision = 4, pipeline } = options;
29
+
30
+ if (!pipeline) pipeline = await getEmbeddingModel();
31
+
32
+ const embedding = await pipeline(text, { pooling: "mean", normalize: true });
33
+
34
+ // Check if input was an array to determine output format
35
+ if (Array.isArray(text)) {
36
+ return embedding
37
+ .tolist()
38
+ .map((vec) => vec.map((num) => parseFloat(num.toFixed(precision))));
39
+ } else {
40
+ const roundedEmbedding = Array.from((embedding as any).data).map(
41
+ (num: any) => parseFloat((num as any).toFixed(precision)),
42
+ );
43
+ return roundedEmbedding;
44
+ }
45
+ }
46
+
47
+ /**
48
+ * Calculate the semantic similarity between one text and a list of
49
+ * other sentences by comparing their embeddings.
50
+ * https://huggingface.co/docs/api-inference/detailed_parameters#sentence-similarity-task
51
+ *
52
+ * <img src="https://i.imgur.com/ex2UWnu.png" width="350px" />
53
+ * @param {string} source_sentence The string that you wish to
54
+ * compare the other strings with. This can be a phrase, sentence,
55
+ * or longer passage, depending on the model being used.
56
+ * @param {Array<string>} sentences A list of strings which will be compared
57
+ * against the source_sentence.
58
+ * @param {Object} [options]
59
+ * @param {string} options.model default="sentence-transformers/all-MiniLM-L6-v2"
60
+ * @param {string} options.HF_API_KEY Required https://huggingface.co/settings/tokens
61
+ * @returns array of 0-1 similarity scores for each sentence
62
+ * @category Similarity
63
+ */
64
+ export async function weighRelevanceConceptVectorAPI(
65
+ source_sentence,
66
+ sentences,
67
+ options = {},
68
+ ) {
69
+ var { model = "sentence-transformers/all-MiniLM-L6-v2", HF_API_KEY = "" } =
70
+ options as any;
71
+
72
+ if (!HF_API_KEY) return { error: "No API key" };
73
+
74
+ const url = `https://api-inference.huggingface.co/models/${model}`;
75
+ try {
76
+ return await grab(url, {
77
+ method: "POST",
78
+ headers: {
79
+ Authorization: `Bearer ${HF_API_KEY}`,
80
+ },
81
+ body: JSON.stringify({
82
+ inputs: {
83
+ source_sentence,
84
+ sentences,
85
+ },
86
+ }),
87
+ });
88
+ } catch (error) {
89
+ console.error("API request failed:", error);
90
+ return null;
91
+ }
92
+ }
93
+
94
+ /**
95
+ * Initialize HuggingFace Transformers pipeline for embedding text.
96
+ *
97
+ * <img src="https://i.imgur.com/3R5Tsrf.png" width="350px" />
98
+ * @param {Object} [options]
99
+ * @param {string} options.pipelineName default "feature-extraction",
100
+ * @param {string} options.modelName default="Xenova/all-MiniLM-L6-v2" -
101
+ * The name of the model to use
102
+ * @returns {Promise<import("@huggingface/transformers").AutoTokenizer>} The pipeline.
103
+ * @category Similarity
104
+ */
105
+ export async function getEmbeddingModel(options: any = {}) {
106
+ const { pipeline } = await (import("@huggingface/transformers") as Promise<any>);
107
+ const {
108
+ pipelineName = "feature-extraction",
109
+ modelName = "Xenova/all-MiniLM-L6-v2",
110
+ quantized = true,
111
+ gpu = false,
112
+ } = options;
113
+
114
+ return await pipeline(pipelineName, modelName, {
115
+ quantized,
116
+ dtype: "fp32",
117
+ device: gpu ? "webgpu" : "cpu",
118
+ } as any);
119
+ }
120
+
121
+ /**
122
+ * [Cosine similarity](https://en.wikipedia.org/wiki/Cosine_similarity) gets similarity of two
123
+ * vectors by whether they have the same direction (similar) or are poles apart. Cosine similarity
124
+ * is often used with text representations to compare how similar two documents or sentences
125
+ * are to each other. The output of cosine similarity ranges from -1 to 1, where -1 means the
126
+ * two vectors are completely dissimilar, and 1 indicates maximum similarity.
127
+ * @param {Array<number>} vectorA
128
+ * @param {Array<number>} vectorB
129
+ * @returns {number} -1 to 1 similarity score
130
+ */
131
+ export function calculateCosineSimilarity(vectorA, vectorB) {
132
+ return (
133
+ vectorA.reduce((sum, a, i) => sum + a * vectorB[i], 0) /
134
+ (Math.sqrt(vectorA.reduce((sum, a) => sum + a * a, 0)) *
135
+ Math.sqrt(vectorB.reduce((sum, b) => sum + b * b, 0)))
136
+ );
137
+ }
138
+
139
+ ///OLDER ======================
140
+ /**
141
+ * Rerank documents's chunks based on relevance to query,
142
+ * based on cosine similarity of their concept vectors generated
143
+ * by a 20MB MiniLM transformer model downloaded locally.
144
+ *
145
+ * [A Complete Overview of Word Embeddings](https://www.youtube.com/watch?v=5MaWmXwxFNQ&t=323s)
146
+ * @param {Array<string>} documents
147
+ * @param {string} query
148
+ * @param {Object} [options]
149
+ * @returns {Promise<Array<{content: string, similarity: number}>>}
150
+ * @category Similarity
151
+ */
152
+ export async function weighRelevanceConceptVector(
153
+ documents,
154
+ query,
155
+ options = {},
156
+ ) {
157
+ const docEmbeddings = await convertTextToEmbedding(documents, options);
158
+
159
+ const queryEmbedding = await convertTextToEmbedding(query, options);
160
+
161
+ let sortedDocs = docEmbeddings
162
+ .map((docEmbedding, i) => ({
163
+ index: i,
164
+ similarity: calculateCosineSimilarity(queryEmbedding, docEmbedding),
165
+ }))
166
+ .sort((a, b) => b.similarity - a.similarity)
167
+ .map(({ index, similarity }) => ({
168
+ content: documents[index],
169
+ similarity,
170
+ }));
171
+
172
+ return sortedDocs;
173
+ }
174
+
175
+ /**
176
+ * Rerank documents's chunks based on relevance to multiple queries,
177
+ * optimizing by embedding documents only once.
178
+ *
179
+ * @param {Array<string>} documents
180
+ * @param {Array<string>} queries
181
+ * @param {Object} [options]
182
+ * @returns {Promise<Object<string, Array<{content: string, similarity: number}>>>}
183
+ * @category Similarity
184
+ */
185
+ export async function weighRelevanceConceptVectorMultiple(
186
+ documents,
187
+ queries,
188
+ options = {},
189
+ ) {
190
+ if (!documents || documents.length === 0 || !queries || queries.length === 0)
191
+ return {};
192
+
193
+ const pipeline = (options as any).pipeline;
194
+ const embedder = pipeline || (await getEmbeddingModel(options));
195
+
196
+ // Embed documents once
197
+ const docEmbeddings = await convertTextToEmbedding(documents, {
198
+ ...options,
199
+ pipeline: embedder,
200
+ });
201
+ // Embed all queries at once
202
+ const queryEmbeddings = await convertTextToEmbedding(queries, {
203
+ ...options,
204
+ pipeline: embedder,
205
+ });
206
+
207
+ // Ensure queryEmbeddings is a 2D array even if queries was a single string wrapped in an array
208
+ const qEmbeds = Array.isArray(queryEmbeddings[0])
209
+ ? queryEmbeddings
210
+ : [queryEmbeddings];
211
+
212
+ const resultsByQuery = {};
213
+
214
+ queries.forEach((query, qIdx) => {
215
+ const queryEmbedding = qEmbeds[qIdx];
216
+
217
+ let sortedDocs = docEmbeddings
218
+ .map((docEmbedding, i) => ({
219
+ index: i,
220
+ similarity: calculateCosineSimilarity(queryEmbedding, docEmbedding),
221
+ }))
222
+ .sort((a, b) => b.similarity - a.similarity)
223
+ .map(({ index, similarity }) => ({
224
+ content: documents[index],
225
+ similarity,
226
+ }));
227
+
228
+ resultsByQuery[query] = sortedDocs;
229
+ });
230
+
231
+ return resultsByQuery;
232
+ }
@@ -0,0 +1,59 @@
1
+ /**
2
+ * @fileoverview Scoring logic for keyphrases based on Wiki-entities, specificity, and query bias.
3
+ * Implements the ranking refinement steps of the SEEKTOPIC pipeline.
4
+ */
5
+ import type { KeyphraseEntry, PhrasesModel } from "./types";
6
+ import { convertTextToTokens } from "../tokenize/text-to-topic-tokens";
7
+
8
+ /**
9
+ * Applies two weighting passes to a keyphrase list:
10
+ *
11
+ * 1. **Wiki-entity bonus** \u2014 if any token in the phrase is a Wikipedia-linked
12
+ * entity (POS tag 5), the keyphrase weight is doubled and `wiki` is set.
13
+ * 2. **IDF domain-specificity** \u2014 weight is multiplied by the average POS
14
+ * uniqueness score across the phrase's tokens. Rare, domain-specific terms
15
+ * (high uniqueness) receive a larger multiplier than common nouns.
16
+ * 3. **Heavy-query bias** (optional) \u2014 if `heavyWeightQuery` is set, keyphrases
17
+ * that closely match the query words receive a large additive bonus, allowing
18
+ * dynamic re-ranking when a user clicks a term or arrives via a search query.
19
+ *
20
+ * @param keyphrases - Keyphrases to weight; mutated in-place for efficiency.
21
+ * @param phrasesModel - Trie model passed to the tokenizer (may be undefined).
22
+ * @param heavyWeightQuery - Space-separated query string to bias ranking toward.
23
+ * @returns The same array with updated `weight` and `wiki` fields.
24
+ *
25
+ * @example
26
+ * const weighted = weightKeyphrasesBySpecificity(keyphrases, model, "self attention");
27
+ */
28
+ export function weightKeyphrasesBySpecificity(
29
+ keyphrases: KeyphraseEntry[],
30
+ phrasesModel: PhrasesModel | undefined,
31
+ heavyWeightQuery: string,
32
+ ): KeyphraseEntry[] {
33
+ const querySplit = heavyWeightQuery ? heavyWeightQuery.split(" ") : [];
34
+
35
+ for (const kp of keyphrases) {
36
+ const tokens = convertTextToTokens(kp.keyphrase, { phrasesModel });
37
+
38
+ // Wiki-entity bonus: any token with category 5 doubles the weight
39
+ if (tokens.some(t => t[1] === 5)) {
40
+ kp.wiki = true;
41
+ kp.weight *= 2;
42
+ }
43
+
44
+ // IDF domain-specificity: scale by average token uniqueness score (index [1])
45
+ // Tokens without a score default to 4 (mid-range common noun)
46
+ const idfAvg = tokens.reduce((sum, t) => sum + (t[1] ?? 4), 0) / tokens.length;
47
+ kp.weight = Math.floor(kp.weight * idfAvg);
48
+
49
+ // Heavy-query bias: boost phrases that closely match the query
50
+ if (querySplit.length) {
51
+ const diffWords = querySplit.filter(w => !kp.keyphrase.includes(w)).length;
52
+ if (diffWords < 4 && diffWords < querySplit.length - 1) {
53
+ kp.weight += 4000 - diffWords * 1000;
54
+ }
55
+ }
56
+ }
57
+
58
+ return keyphrases;
59
+ }