extract-webpage 1.2.34 → 1.2.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,279 +1,279 @@
1
- /**
2
- * @fileoverview Main orchestrator for the SEEKTOPIC keyphrase extraction pipeline.
3
- * Coordinates cleaning, segmentation, LLM extraction, and fallback n-gram ranking.
4
- */
5
- import type {
6
- NgramMap,
7
- KeyphraseEntry,
8
- SentenceEntry,
9
- SEEKTOPICOptions,
10
- SEEKTOPICResult,
11
- } from "./types";
12
- import { splitTextToSentences } from "../tokenize/text-to-sentences";
13
- import { convertTextToTokens } from "../tokenize/text-to-topic-tokens";
14
- import { extractNounEdgeGrams } from "./ngrams";
15
- import { foldSubphrases } from "./fold-keyphrases";
16
- import { weightKeyphrasesBySpecificity } from "./weight-keyphrases";
17
- import { rankSentencesCentralToKeyphrase } from "./rank-sentences-keyphrases";
18
- import { weighRelevanceConceptVectorMultiple } from "./vector-search";
19
-
20
- /**
21
- * ### SEEKTOPIC \u2014 Keyphrase & Sentence Extraction
22
- *
23
- * Pulls the most important phrases and sentences out of any document.
24
- * Given raw text, it returns a ranked list of key concepts (e.g. "neural
25
- * network", "climate change") and, optionally, the sentences that best
26
- * summarise the document around those concepts.
27
- *
28
- * <img src="https://i.imgur.com/gZ4kI1V.png" width="360px" />
29
- *
30
- * **How it works \u2014 8-step pipeline:**
31
- *
32
- * 1. **Clean** \u2014 strip HTML tags and entities.
33
- * 2. **Split** \u2014 break text into sentences, respecting abbreviations and URLs.
34
- * 3. **Topic Extraction** *(LLM Path)* \u2014 sends the first 5000 words to an LLM
35
- * to extract the most descriptive key topic phrases labeling the document.
36
- * 4. **Vector Search** *(LLM Path)* \u2014 calculates cosine similarity embeddings
37
- * for all sentences vs topic phrases, finding the top relevant sentences per topic.
38
- * 5. **Tokenise & Extract** *(Fallback)* \u2014 if LLM fails, identify noun-anchored
39
- * n-grams (1-N words).
40
- * 6. **Score & Fold** *(Fallback)* \u2014 weight phrases and merge subsumed phrases.
41
- * 7. **Re-weight** *(Fallback)* \u2014 boost Wikipedia entities and rare domain terms.
42
- * 8. **TextRank** *(Fallback)* \u2014 build sentence similarity graph and run random walk.
43
- *
44
- * <video src="https://github.com/user-attachments/assets/73348d63-7671-4e20-8df9-29a13d5b0768"
45
- * width="550px" controls />
46
- *
47
- * @param docText - Plain text or HTML document to analyse.
48
- * @param options - Optional tuning parameters (word limits, thresholds, query bias).
49
- * @returns
50
- * - `optionSkipRanking: true` (default) \u2192 `KeyphraseEntry[]` sorted by weight.
51
- * - `optionSkipRanking: false` \u2192 `SEEKTOPICResult` with `topSentences`,
52
- * `keyphrases`, and the full `sentences` array.
53
- *
54
- * @example
55
- * // Fast: just extract keyphrases
56
- * const keyphrases = extractSEEKTOPIC(articleText, { phrasesModel });
57
- * // \u2192 [{ keyphrase: "machine learning", weight: 84 }, ...]
58
- *
59
- * @example
60
- * // Full: keyphrases + summary sentences, biased toward a search query
61
- * const { topSentences, keyphrases } = extractSEEKTOPIC(articleText, {
62
- * phrasesModel,
63
- * optionSkipRanking: false,
64
- * heavyWeightQuery: "transformer attention",
65
- * limitTopSentences: 5,
66
- * }) as SEEKTOPICResult;
67
- *
68
- * @author [ai-research-agent (2024)](https://airesearch.js.org)
69
- * @category Topics
70
- */
71
- export async function extractSEEKTOPIC(
72
- docText: string,
73
- options: SEEKTOPICOptions = {},
74
- ): Promise<KeyphraseEntry[] | SEEKTOPICResult> {
75
- if (typeof docText !== "string") throw new Error("docText must be a string");
76
-
77
- const {
78
- phrasesModel,
79
- maxWords = 2,
80
- minWords = 1,
81
- minWordLength = 3,
82
- topKeyphrasesPercent = 0.5,
83
- limitTopSentences = 5,
84
- limitTopKeyphrases = 10,
85
- minKeyPhraseLength = 5,
86
- heavyWeightQuery = "",
87
- removeHTML = true,
88
- optionSkipRanking = true,
89
- getEnv = () => "",
90
- } = options;
91
-
92
- // \u2500\u2500 1. Normalize HTML \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
93
- let text = docText
94
- .replace(/</g, " <")
95
- .replace(/>/g, "> ")
96
- .replace(/&.{2,5};/g, ""); // strip &quot; &amp; &lt; &gt; &nbsp;
97
- if (removeHTML) text = text.replace(/<[^>]*>/g, "");
98
-
99
- // \u2500\u2500 2. Sentence segmentation \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
100
- const sentencesArray = splitTextToSentences(text);
101
-
102
- // \u2500\u2500 3. LLM Topic Extraction (Starting Step) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
103
- const first5kWords = text.split(/\s+/).slice(0, 5000).join(" ");
104
- const llmPrompt = `Extract the 5-10 most important key topic phrases that best
105
- label this document. Return exclusively a comma-separated list of the key topic
106
- phrases. Document text: ${first5kWords}`;
107
-
108
- let llmTopics: string[] = [];
109
- try {
110
- // `generate-language` moved into the chat-agent-toolkit workspace package.
111
- // Load it lazily (and tolerate absence) since the LLM path is an optional
112
- // enhancement — the n-gram fallback below covers extraction without it.
113
- const { writeLanguageResponse } = await import("chat-agent-toolkit");
114
- const aiResponse = await writeLanguageResponse({
115
- query: llmPrompt,
116
- provider: getEnv("LLM_PROVIDER") || "openai",
117
- apiKey: getEnv("LLM_API_KEY") || getEnv("OPENAI_API_KEY") || "dummy",
118
- });
119
- if (aiResponse.content && !aiResponse.error) {
120
- llmTopics = aiResponse.content
121
- .split(",")
122
- .map((s: string) => s.trim().replace(/^['"]|['"]$/g, ""))
123
- .filter(Boolean);
124
- }
125
- } catch (e) {
126
- console.error("LLM topic extraction failed", e);
127
- }
128
-
129
- // \u2500\u2500 4. Use vector-search.ts for top topic sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
130
- if (llmTopics.length > 0) {
131
- if (optionSkipRanking) {
132
- return llmTopics.map((topic: string) => ({
133
- keyphrase: topic,
134
- sentences: [],
135
- words: topic.split(/\s+/).length,
136
- weight: 100,
137
- }));
138
- }
139
-
140
- const { limitTopSentences = 3 } = options;
141
- const allRelevant = await weighRelevanceConceptVectorMultiple(
142
- sentencesArray,
143
- llmTopics,
144
- options,
145
- );
146
-
147
- const keyphraseObjects: any[] = [];
148
- const topSentences: any[] = [];
149
-
150
- for (const topic of llmTopics) {
151
- // @ts-ignore
152
- const topicRelevant = allRelevant[topic] || [];
153
- const topX = topicRelevant.slice(0, limitTopSentences);
154
-
155
- const topicSentences = topX.map((r: any) => ({
156
- text: r.content,
157
- similarity: r.similarity,
158
- }));
159
-
160
- keyphraseObjects.push({
161
- keyphrase: topic,
162
- topSentences: topicSentences,
163
- words: topic.split(/\s+/).length,
164
- weight: 100,
165
- });
166
-
167
- for (const r of topX) {
168
- const sIdx = sentencesArray.indexOf(r.content);
169
- topSentences.push({
170
- text: r.content,
171
- index: sIdx !== -1 ? sIdx : 0,
172
- keyphrases: [{ keyphrase: topic, weight: r.similarity }],
173
- weight: r.similarity,
174
- });
175
- }
176
- }
177
-
178
- // Deduplicate topSentences
179
- const uniqueSentences = Array.from(
180
- new Map(topSentences.map((s: any) => [s.text, s])).values(),
181
- );
182
-
183
- return {
184
- topSentences: uniqueSentences.sort(
185
- (a: any, b: any) => b.weight - a.weight,
186
- ),
187
- keyphrases: keyphraseObjects,
188
- sentences: sentencesArray,
189
- } as unknown as SEEKTOPICResult;
190
- }
191
-
192
- // \u2500\u2500 FALLBACK: Tokenise + extract n-grams \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
193
- const nGrams: NgramMap = {};
194
- const sentenceKeysMap: SentenceEntry[] = [];
195
-
196
- for (let idx = 0; idx < sentencesArray.length; idx++) {
197
- const sentence = sentencesArray[idx];
198
- const tokens = convertTextToTokens(sentence, { phrasesModel });
199
-
200
- sentenceKeysMap.push({
201
- text: sentence,
202
- index: idx,
203
- keyphrases: [],
204
- weight: 0,
205
- });
206
-
207
- for (let i = 0; i < tokens.length; i++) {
208
- for (let n = minWords; n <= maxWords; n++) {
209
- extractNounEdgeGrams(n, tokens, i, nGrams, minWordLength, idx);
210
- }
211
- }
212
- }
213
-
214
- // \u2500\u2500 5. Score raw keyphrases \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
215
- let keyphraseObjects: KeyphraseEntry[] = [];
216
- for (let n = minWords; n <= maxWords; n++) {
217
- if (!nGrams[n]) continue;
218
- for (const [keyphrase, sentencesArr] of Object.entries(nGrams[n])) {
219
- keyphraseObjects.push({
220
- keyphrase,
221
- sentences: sentencesArr,
222
- words: n,
223
- weight: sentencesArr.length * n,
224
- });
225
- }
226
- }
227
-
228
- // \u2500\u2500 6. Fold subphrases + deduplicate \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
229
- const folded = foldSubphrases(keyphraseObjects);
230
- const seen = new Set<string>();
231
- keyphraseObjects = folded
232
- .sort((a, b) => b.weight - a.weight)
233
- .filter((k) => !seen.has(k.keyphrase) && seen.add(k.keyphrase))
234
- .map((k) => ({
235
- ...k,
236
- sentences: Array.isArray(k.sentences)
237
- ? [...new Set(k.sentences as number[])]
238
- : k.sentences,
239
- }));
240
-
241
- // \u2500\u2500 7. IDF + wiki-entity + heavy-query weighting \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
242
- keyphraseObjects = weightKeyphrasesBySpecificity(
243
- keyphraseObjects,
244
- phrasesModel,
245
- heavyWeightQuery,
246
- )
247
- .filter((k) => k.keyphrase.length > minKeyPhraseLength)
248
- .sort((a, b) => b.weight - a.weight);
249
-
250
- // Fast path: return keyphrases without TextRank
251
- if (optionSkipRanking) return keyphraseObjects;
252
-
253
- // \u2500\u2500 8. Attach keyphrases to sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
254
- for (const { keyphrase, sentences, weight } of keyphraseObjects) {
255
- if (Array.isArray(sentences)) {
256
- for (const si of sentences) {
257
- sentenceKeysMap[si as number]?.keyphrases.push({ keyphrase, weight });
258
- }
259
- }
260
- }
261
-
262
- // \u2500\u2500 9. TextRank \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
263
- const ranked = rankSentencesCentralToKeyphrase(sentenceKeysMap);
264
-
265
- const topSentences = (ranked ?? [])
266
- .sort((a, b) => b.weight - a.weight)
267
- .slice(0, limitTopSentences)
268
- .map((s) => ({
269
- ...s,
270
- keyphrases: [...new Set(s.keyphrases.map((k) => k.keyphrase))],
271
- }));
272
-
273
- const keyphrases = keyphraseObjects.slice(0, limitTopKeyphrases).map((k) => ({
274
- ...k,
275
- sentences: Array.isArray(k.sentences) ? k.sentences.join(",") : k.sentences,
276
- }));
277
-
278
- return { topSentences, keyphrases, sentences: sentencesArray };
279
- }
1
+ /**
2
+ * @fileoverview Main orchestrator for the SEEKTOPIC keyphrase extraction pipeline.
3
+ * Coordinates cleaning, segmentation, LLM extraction, and fallback n-gram ranking.
4
+ */
5
+ import type {
6
+ NgramMap,
7
+ KeyphraseEntry,
8
+ SentenceEntry,
9
+ SEEKTOPICOptions,
10
+ SEEKTOPICResult,
11
+ } from "./types";
12
+ import { splitTextToSentences } from "../tokenize/text-to-sentences";
13
+ import { convertTextToTokens } from "../tokenize/text-to-topic-tokens";
14
+ import { extractNounEdgeGrams } from "./ngrams";
15
+ import { foldSubphrases } from "./fold-keyphrases";
16
+ import { weightKeyphrasesBySpecificity } from "./weight-keyphrases";
17
+ import { rankSentencesCentralToKeyphrase } from "./rank-sentences-keyphrases";
18
+ import { weighRelevanceConceptVectorMultiple } from "./vector-search";
19
+
20
+ /**
21
+ * ### SEEKTOPIC \u2014 Keyphrase & Sentence Extraction
22
+ *
23
+ * Pulls the most important phrases and sentences out of any document.
24
+ * Given raw text, it returns a ranked list of key concepts (e.g. "neural
25
+ * network", "climate change") and, optionally, the sentences that best
26
+ * summarise the document around those concepts.
27
+ *
28
+ * <img src="https://i.imgur.com/gZ4kI1V.png" width="360px" />
29
+ *
30
+ * **How it works \u2014 8-step pipeline:**
31
+ *
32
+ * 1. **Clean** \u2014 strip HTML tags and entities.
33
+ * 2. **Split** \u2014 break text into sentences, respecting abbreviations and URLs.
34
+ * 3. **Topic Extraction** *(LLM Path)* \u2014 sends the first 5000 words to an LLM
35
+ * to extract the most descriptive key topic phrases labeling the document.
36
+ * 4. **Vector Search** *(LLM Path)* \u2014 calculates cosine similarity embeddings
37
+ * for all sentences vs topic phrases, finding the top relevant sentences per topic.
38
+ * 5. **Tokenise & Extract** *(Fallback)* \u2014 if LLM fails, identify noun-anchored
39
+ * n-grams (1-N words).
40
+ * 6. **Score & Fold** *(Fallback)* \u2014 weight phrases and merge subsumed phrases.
41
+ * 7. **Re-weight** *(Fallback)* \u2014 boost Wikipedia entities and rare domain terms.
42
+ * 8. **TextRank** *(Fallback)* \u2014 build sentence similarity graph and run random walk.
43
+ *
44
+ * <video src="https://github.com/user-attachments/assets/73348d63-7671-4e20-8df9-29a13d5b0768"
45
+ * width="550px" controls />
46
+ *
47
+ * @param docText - Plain text or HTML document to analyse.
48
+ * @param options - Optional tuning parameters (word limits, thresholds, query bias).
49
+ * @returns
50
+ * - `optionSkipRanking: true` (default) \u2192 `KeyphraseEntry[]` sorted by weight.
51
+ * - `optionSkipRanking: false` \u2192 `SEEKTOPICResult` with `topSentences`,
52
+ * `keyphrases`, and the full `sentences` array.
53
+ *
54
+ * @example
55
+ * // Fast: just extract keyphrases
56
+ * const keyphrases = extractSEEKTOPIC(articleText, { phrasesModel });
57
+ * // \u2192 [{ keyphrase: "machine learning", weight: 84 }, ...]
58
+ *
59
+ * @example
60
+ * // Full: keyphrases + summary sentences, biased toward a search query
61
+ * const { topSentences, keyphrases } = extractSEEKTOPIC(articleText, {
62
+ * phrasesModel,
63
+ * optionSkipRanking: false,
64
+ * heavyWeightQuery: "transformer attention",
65
+ * limitTopSentences: 5,
66
+ * }) as SEEKTOPICResult;
67
+ *
68
+ * @author [ai-research-agent (2024)](https://airesearch.js.org)
69
+ * @category Topics
70
+ */
71
+ export async function extractSEEKTOPIC(
72
+ docText: string,
73
+ options: SEEKTOPICOptions = {},
74
+ ): Promise<KeyphraseEntry[] | SEEKTOPICResult> {
75
+ if (typeof docText !== "string") throw new Error("docText must be a string");
76
+
77
+ const {
78
+ phrasesModel,
79
+ maxWords = 2,
80
+ minWords = 1,
81
+ minWordLength = 3,
82
+ topKeyphrasesPercent = 0.5,
83
+ limitTopSentences = 5,
84
+ limitTopKeyphrases = 10,
85
+ minKeyPhraseLength = 5,
86
+ heavyWeightQuery = "",
87
+ removeHTML = true,
88
+ optionSkipRanking = true,
89
+ getEnv = () => "",
90
+ } = options;
91
+
92
+ // \u2500\u2500 1. Normalize HTML \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
93
+ let text = docText
94
+ .replace(/</g, " <")
95
+ .replace(/>/g, "> ")
96
+ .replace(/&.{2,5};/g, ""); // strip &quot; &amp; &lt; &gt; &nbsp;
97
+ if (removeHTML) text = text.replace(/<[^>]*>/g, "");
98
+
99
+ // \u2500\u2500 2. Sentence segmentation \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
100
+ const sentencesArray = splitTextToSentences(text);
101
+
102
+ // \u2500\u2500 3. LLM Topic Extraction (Starting Step) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
103
+ const first5kWords = text.split(/\s+/).slice(0, 5000).join(" ");
104
+ const llmPrompt = `Extract the 5-10 most important key topic phrases that best
105
+ label this document. Return exclusively a comma-separated list of the key topic
106
+ phrases. Document text: ${first5kWords}`;
107
+
108
+ let llmTopics: string[] = [];
109
+ try {
110
+ // `generate-language` moved into the chat-agent-toolkit workspace package.
111
+ // Load it lazily (and tolerate absence) since the LLM path is an optional
112
+ // enhancement — the n-gram fallback below covers extraction without it.
113
+ const { writeLanguageResponse } = await import("chat-agent-toolkit");
114
+ const aiResponse = await writeLanguageResponse({
115
+ query: llmPrompt,
116
+ provider: getEnv("LLM_PROVIDER") || "openai",
117
+ apiKey: getEnv("LLM_API_KEY") || getEnv("OPENAI_API_KEY") || "dummy",
118
+ });
119
+ if (aiResponse.content && !aiResponse.error) {
120
+ llmTopics = aiResponse.content
121
+ .split(",")
122
+ .map((s: string) => s.trim().replace(/^['"]|['"]$/g, ""))
123
+ .filter(Boolean);
124
+ }
125
+ } catch (e) {
126
+ console.error("LLM topic extraction failed", e);
127
+ }
128
+
129
+ // \u2500\u2500 4. Use vector-search.ts for top topic sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
130
+ if (llmTopics.length > 0) {
131
+ if (optionSkipRanking) {
132
+ return llmTopics.map((topic: string) => ({
133
+ keyphrase: topic,
134
+ sentences: [],
135
+ words: topic.split(/\s+/).length,
136
+ weight: 100,
137
+ }));
138
+ }
139
+
140
+ const { limitTopSentences = 3 } = options;
141
+ const allRelevant = await weighRelevanceConceptVectorMultiple(
142
+ sentencesArray,
143
+ llmTopics,
144
+ options,
145
+ );
146
+
147
+ const keyphraseObjects: any[] = [];
148
+ const topSentences: any[] = [];
149
+
150
+ for (const topic of llmTopics) {
151
+ // @ts-ignore
152
+ const topicRelevant = allRelevant[topic] || [];
153
+ const topX = topicRelevant.slice(0, limitTopSentences);
154
+
155
+ const topicSentences = topX.map((r: any) => ({
156
+ text: r.content,
157
+ similarity: r.similarity,
158
+ }));
159
+
160
+ keyphraseObjects.push({
161
+ keyphrase: topic,
162
+ topSentences: topicSentences,
163
+ words: topic.split(/\s+/).length,
164
+ weight: 100,
165
+ });
166
+
167
+ for (const r of topX) {
168
+ const sIdx = sentencesArray.indexOf(r.content);
169
+ topSentences.push({
170
+ text: r.content,
171
+ index: sIdx !== -1 ? sIdx : 0,
172
+ keyphrases: [{ keyphrase: topic, weight: r.similarity }],
173
+ weight: r.similarity,
174
+ });
175
+ }
176
+ }
177
+
178
+ // Deduplicate topSentences
179
+ const uniqueSentences = Array.from(
180
+ new Map(topSentences.map((s: any) => [s.text, s])).values(),
181
+ );
182
+
183
+ return {
184
+ topSentences: uniqueSentences.sort(
185
+ (a: any, b: any) => b.weight - a.weight,
186
+ ),
187
+ keyphrases: keyphraseObjects,
188
+ sentences: sentencesArray,
189
+ } as unknown as SEEKTOPICResult;
190
+ }
191
+
192
+ // \u2500\u2500 FALLBACK: Tokenise + extract n-grams \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
193
+ const nGrams: NgramMap = {};
194
+ const sentenceKeysMap: SentenceEntry[] = [];
195
+
196
+ for (let idx = 0; idx < sentencesArray.length; idx++) {
197
+ const sentence = sentencesArray[idx];
198
+ const tokens = convertTextToTokens(sentence, { phrasesModel });
199
+
200
+ sentenceKeysMap.push({
201
+ text: sentence,
202
+ index: idx,
203
+ keyphrases: [],
204
+ weight: 0,
205
+ });
206
+
207
+ for (let i = 0; i < tokens.length; i++) {
208
+ for (let n = minWords; n <= maxWords; n++) {
209
+ extractNounEdgeGrams(n, tokens, i, nGrams, minWordLength, idx);
210
+ }
211
+ }
212
+ }
213
+
214
+ // \u2500\u2500 5. Score raw keyphrases \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
215
+ let keyphraseObjects: KeyphraseEntry[] = [];
216
+ for (let n = minWords; n <= maxWords; n++) {
217
+ if (!nGrams[n]) continue;
218
+ for (const [keyphrase, sentencesArr] of Object.entries(nGrams[n])) {
219
+ keyphraseObjects.push({
220
+ keyphrase,
221
+ sentences: sentencesArr,
222
+ words: n,
223
+ weight: sentencesArr.length * n,
224
+ });
225
+ }
226
+ }
227
+
228
+ // \u2500\u2500 6. Fold subphrases + deduplicate \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
229
+ const folded = foldSubphrases(keyphraseObjects);
230
+ const seen = new Set<string>();
231
+ keyphraseObjects = folded
232
+ .sort((a, b) => b.weight - a.weight)
233
+ .filter((k) => !seen.has(k.keyphrase) && seen.add(k.keyphrase))
234
+ .map((k) => ({
235
+ ...k,
236
+ sentences: Array.isArray(k.sentences)
237
+ ? [...new Set(k.sentences as number[])]
238
+ : k.sentences,
239
+ }));
240
+
241
+ // \u2500\u2500 7. IDF + wiki-entity + heavy-query weighting \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
242
+ keyphraseObjects = weightKeyphrasesBySpecificity(
243
+ keyphraseObjects,
244
+ phrasesModel,
245
+ heavyWeightQuery,
246
+ )
247
+ .filter((k) => k.keyphrase.length > minKeyPhraseLength)
248
+ .sort((a, b) => b.weight - a.weight);
249
+
250
+ // Fast path: return keyphrases without TextRank
251
+ if (optionSkipRanking) return keyphraseObjects;
252
+
253
+ // \u2500\u2500 8. Attach keyphrases to sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
254
+ for (const { keyphrase, sentences, weight } of keyphraseObjects) {
255
+ if (Array.isArray(sentences)) {
256
+ for (const si of sentences) {
257
+ sentenceKeysMap[si as number]?.keyphrases.push({ keyphrase, weight });
258
+ }
259
+ }
260
+ }
261
+
262
+ // \u2500\u2500 9. TextRank \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
263
+ const ranked = rankSentencesCentralToKeyphrase(sentenceKeysMap);
264
+
265
+ const topSentences = (ranked ?? [])
266
+ .sort((a, b) => b.weight - a.weight)
267
+ .slice(0, limitTopSentences)
268
+ .map((s) => ({
269
+ ...s,
270
+ keyphrases: [...new Set(s.keyphrases.map((k) => k.keyphrase))],
271
+ }));
272
+
273
+ const keyphrases = keyphraseObjects.slice(0, limitTopKeyphrases).map((k) => ({
274
+ ...k,
275
+ sentences: Array.isArray(k.sentences) ? k.sentences.join(",") : k.sentences,
276
+ }));
277
+
278
+ return { topSentences, keyphrases, sentences: sentencesArray };
279
+ }