extract-webpage 1.2.34 → 1.2.36
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js.map +1 -1
- package/package.json +4 -4
- package/src/fs-mock.js +22 -22
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -1049
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3452 -3452
- package/src/search/index.ts +45 -45
- package/src/search/meta-search-agent-reexport.ts +38 -38
- package/src/search/url-to-html.ts +278 -278
- package/src/seektopic/seektopic-keyphrases.ts +279 -279
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -435
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -332
- package/src/url-to-content/url-to-content.ts +367 -367
- package/src/url-to-content/url-to-html.ts +436 -436
- package/src/utils/grab.ts +51 -51
|
@@ -1,279 +1,279 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @fileoverview Main orchestrator for the SEEKTOPIC keyphrase extraction pipeline.
|
|
3
|
-
* Coordinates cleaning, segmentation, LLM extraction, and fallback n-gram ranking.
|
|
4
|
-
*/
|
|
5
|
-
import type {
|
|
6
|
-
NgramMap,
|
|
7
|
-
KeyphraseEntry,
|
|
8
|
-
SentenceEntry,
|
|
9
|
-
SEEKTOPICOptions,
|
|
10
|
-
SEEKTOPICResult,
|
|
11
|
-
} from "./types";
|
|
12
|
-
import { splitTextToSentences } from "../tokenize/text-to-sentences";
|
|
13
|
-
import { convertTextToTokens } from "../tokenize/text-to-topic-tokens";
|
|
14
|
-
import { extractNounEdgeGrams } from "./ngrams";
|
|
15
|
-
import { foldSubphrases } from "./fold-keyphrases";
|
|
16
|
-
import { weightKeyphrasesBySpecificity } from "./weight-keyphrases";
|
|
17
|
-
import { rankSentencesCentralToKeyphrase } from "./rank-sentences-keyphrases";
|
|
18
|
-
import { weighRelevanceConceptVectorMultiple } from "./vector-search";
|
|
19
|
-
|
|
20
|
-
/**
|
|
21
|
-
* ### SEEKTOPIC \u2014 Keyphrase & Sentence Extraction
|
|
22
|
-
*
|
|
23
|
-
* Pulls the most important phrases and sentences out of any document.
|
|
24
|
-
* Given raw text, it returns a ranked list of key concepts (e.g. "neural
|
|
25
|
-
* network", "climate change") and, optionally, the sentences that best
|
|
26
|
-
* summarise the document around those concepts.
|
|
27
|
-
*
|
|
28
|
-
* <img src="https://i.imgur.com/gZ4kI1V.png" width="360px" />
|
|
29
|
-
*
|
|
30
|
-
* **How it works \u2014 8-step pipeline:**
|
|
31
|
-
*
|
|
32
|
-
* 1. **Clean** \u2014 strip HTML tags and entities.
|
|
33
|
-
* 2. **Split** \u2014 break text into sentences, respecting abbreviations and URLs.
|
|
34
|
-
* 3. **Topic Extraction** *(LLM Path)* \u2014 sends the first 5000 words to an LLM
|
|
35
|
-
* to extract the most descriptive key topic phrases labeling the document.
|
|
36
|
-
* 4. **Vector Search** *(LLM Path)* \u2014 calculates cosine similarity embeddings
|
|
37
|
-
* for all sentences vs topic phrases, finding the top relevant sentences per topic.
|
|
38
|
-
* 5. **Tokenise & Extract** *(Fallback)* \u2014 if LLM fails, identify noun-anchored
|
|
39
|
-
* n-grams (1-N words).
|
|
40
|
-
* 6. **Score & Fold** *(Fallback)* \u2014 weight phrases and merge subsumed phrases.
|
|
41
|
-
* 7. **Re-weight** *(Fallback)* \u2014 boost Wikipedia entities and rare domain terms.
|
|
42
|
-
* 8. **TextRank** *(Fallback)* \u2014 build sentence similarity graph and run random walk.
|
|
43
|
-
*
|
|
44
|
-
* <video src="https://github.com/user-attachments/assets/73348d63-7671-4e20-8df9-29a13d5b0768"
|
|
45
|
-
* width="550px" controls />
|
|
46
|
-
*
|
|
47
|
-
* @param docText - Plain text or HTML document to analyse.
|
|
48
|
-
* @param options - Optional tuning parameters (word limits, thresholds, query bias).
|
|
49
|
-
* @returns
|
|
50
|
-
* - `optionSkipRanking: true` (default) \u2192 `KeyphraseEntry[]` sorted by weight.
|
|
51
|
-
* - `optionSkipRanking: false` \u2192 `SEEKTOPICResult` with `topSentences`,
|
|
52
|
-
* `keyphrases`, and the full `sentences` array.
|
|
53
|
-
*
|
|
54
|
-
* @example
|
|
55
|
-
* // Fast: just extract keyphrases
|
|
56
|
-
* const keyphrases = extractSEEKTOPIC(articleText, { phrasesModel });
|
|
57
|
-
* // \u2192 [{ keyphrase: "machine learning", weight: 84 }, ...]
|
|
58
|
-
*
|
|
59
|
-
* @example
|
|
60
|
-
* // Full: keyphrases + summary sentences, biased toward a search query
|
|
61
|
-
* const { topSentences, keyphrases } = extractSEEKTOPIC(articleText, {
|
|
62
|
-
* phrasesModel,
|
|
63
|
-
* optionSkipRanking: false,
|
|
64
|
-
* heavyWeightQuery: "transformer attention",
|
|
65
|
-
* limitTopSentences: 5,
|
|
66
|
-
* }) as SEEKTOPICResult;
|
|
67
|
-
*
|
|
68
|
-
* @author [ai-research-agent (2024)](https://airesearch.js.org)
|
|
69
|
-
* @category Topics
|
|
70
|
-
*/
|
|
71
|
-
export async function extractSEEKTOPIC(
|
|
72
|
-
docText: string,
|
|
73
|
-
options: SEEKTOPICOptions = {},
|
|
74
|
-
): Promise<KeyphraseEntry[] | SEEKTOPICResult> {
|
|
75
|
-
if (typeof docText !== "string") throw new Error("docText must be a string");
|
|
76
|
-
|
|
77
|
-
const {
|
|
78
|
-
phrasesModel,
|
|
79
|
-
maxWords = 2,
|
|
80
|
-
minWords = 1,
|
|
81
|
-
minWordLength = 3,
|
|
82
|
-
topKeyphrasesPercent = 0.5,
|
|
83
|
-
limitTopSentences = 5,
|
|
84
|
-
limitTopKeyphrases = 10,
|
|
85
|
-
minKeyPhraseLength = 5,
|
|
86
|
-
heavyWeightQuery = "",
|
|
87
|
-
removeHTML = true,
|
|
88
|
-
optionSkipRanking = true,
|
|
89
|
-
getEnv = () => "",
|
|
90
|
-
} = options;
|
|
91
|
-
|
|
92
|
-
// \u2500\u2500 1. Normalize HTML \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
93
|
-
let text = docText
|
|
94
|
-
.replace(/</g, " <")
|
|
95
|
-
.replace(/>/g, "> ")
|
|
96
|
-
.replace(/&.{2,5};/g, ""); // strip " & < >
|
|
97
|
-
if (removeHTML) text = text.replace(/<[^>]*>/g, "");
|
|
98
|
-
|
|
99
|
-
// \u2500\u2500 2. Sentence segmentation \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
100
|
-
const sentencesArray = splitTextToSentences(text);
|
|
101
|
-
|
|
102
|
-
// \u2500\u2500 3. LLM Topic Extraction (Starting Step) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
103
|
-
const first5kWords = text.split(/\s+/).slice(0, 5000).join(" ");
|
|
104
|
-
const llmPrompt = `Extract the 5-10 most important key topic phrases that best
|
|
105
|
-
label this document. Return exclusively a comma-separated list of the key topic
|
|
106
|
-
phrases. Document text: ${first5kWords}`;
|
|
107
|
-
|
|
108
|
-
let llmTopics: string[] = [];
|
|
109
|
-
try {
|
|
110
|
-
// `generate-language` moved into the chat-agent-toolkit workspace package.
|
|
111
|
-
// Load it lazily (and tolerate absence) since the LLM path is an optional
|
|
112
|
-
// enhancement — the n-gram fallback below covers extraction without it.
|
|
113
|
-
const { writeLanguageResponse } = await import("chat-agent-toolkit");
|
|
114
|
-
const aiResponse = await writeLanguageResponse({
|
|
115
|
-
query: llmPrompt,
|
|
116
|
-
provider: getEnv("LLM_PROVIDER") || "openai",
|
|
117
|
-
apiKey: getEnv("LLM_API_KEY") || getEnv("OPENAI_API_KEY") || "dummy",
|
|
118
|
-
});
|
|
119
|
-
if (aiResponse.content && !aiResponse.error) {
|
|
120
|
-
llmTopics = aiResponse.content
|
|
121
|
-
.split(",")
|
|
122
|
-
.map((s: string) => s.trim().replace(/^['"]|['"]$/g, ""))
|
|
123
|
-
.filter(Boolean);
|
|
124
|
-
}
|
|
125
|
-
} catch (e) {
|
|
126
|
-
console.error("LLM topic extraction failed", e);
|
|
127
|
-
}
|
|
128
|
-
|
|
129
|
-
// \u2500\u2500 4. Use vector-search.ts for top topic sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
130
|
-
if (llmTopics.length > 0) {
|
|
131
|
-
if (optionSkipRanking) {
|
|
132
|
-
return llmTopics.map((topic: string) => ({
|
|
133
|
-
keyphrase: topic,
|
|
134
|
-
sentences: [],
|
|
135
|
-
words: topic.split(/\s+/).length,
|
|
136
|
-
weight: 100,
|
|
137
|
-
}));
|
|
138
|
-
}
|
|
139
|
-
|
|
140
|
-
const { limitTopSentences = 3 } = options;
|
|
141
|
-
const allRelevant = await weighRelevanceConceptVectorMultiple(
|
|
142
|
-
sentencesArray,
|
|
143
|
-
llmTopics,
|
|
144
|
-
options,
|
|
145
|
-
);
|
|
146
|
-
|
|
147
|
-
const keyphraseObjects: any[] = [];
|
|
148
|
-
const topSentences: any[] = [];
|
|
149
|
-
|
|
150
|
-
for (const topic of llmTopics) {
|
|
151
|
-
// @ts-ignore
|
|
152
|
-
const topicRelevant = allRelevant[topic] || [];
|
|
153
|
-
const topX = topicRelevant.slice(0, limitTopSentences);
|
|
154
|
-
|
|
155
|
-
const topicSentences = topX.map((r: any) => ({
|
|
156
|
-
text: r.content,
|
|
157
|
-
similarity: r.similarity,
|
|
158
|
-
}));
|
|
159
|
-
|
|
160
|
-
keyphraseObjects.push({
|
|
161
|
-
keyphrase: topic,
|
|
162
|
-
topSentences: topicSentences,
|
|
163
|
-
words: topic.split(/\s+/).length,
|
|
164
|
-
weight: 100,
|
|
165
|
-
});
|
|
166
|
-
|
|
167
|
-
for (const r of topX) {
|
|
168
|
-
const sIdx = sentencesArray.indexOf(r.content);
|
|
169
|
-
topSentences.push({
|
|
170
|
-
text: r.content,
|
|
171
|
-
index: sIdx !== -1 ? sIdx : 0,
|
|
172
|
-
keyphrases: [{ keyphrase: topic, weight: r.similarity }],
|
|
173
|
-
weight: r.similarity,
|
|
174
|
-
});
|
|
175
|
-
}
|
|
176
|
-
}
|
|
177
|
-
|
|
178
|
-
// Deduplicate topSentences
|
|
179
|
-
const uniqueSentences = Array.from(
|
|
180
|
-
new Map(topSentences.map((s: any) => [s.text, s])).values(),
|
|
181
|
-
);
|
|
182
|
-
|
|
183
|
-
return {
|
|
184
|
-
topSentences: uniqueSentences.sort(
|
|
185
|
-
(a: any, b: any) => b.weight - a.weight,
|
|
186
|
-
),
|
|
187
|
-
keyphrases: keyphraseObjects,
|
|
188
|
-
sentences: sentencesArray,
|
|
189
|
-
} as unknown as SEEKTOPICResult;
|
|
190
|
-
}
|
|
191
|
-
|
|
192
|
-
// \u2500\u2500 FALLBACK: Tokenise + extract n-grams \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
193
|
-
const nGrams: NgramMap = {};
|
|
194
|
-
const sentenceKeysMap: SentenceEntry[] = [];
|
|
195
|
-
|
|
196
|
-
for (let idx = 0; idx < sentencesArray.length; idx++) {
|
|
197
|
-
const sentence = sentencesArray[idx];
|
|
198
|
-
const tokens = convertTextToTokens(sentence, { phrasesModel });
|
|
199
|
-
|
|
200
|
-
sentenceKeysMap.push({
|
|
201
|
-
text: sentence,
|
|
202
|
-
index: idx,
|
|
203
|
-
keyphrases: [],
|
|
204
|
-
weight: 0,
|
|
205
|
-
});
|
|
206
|
-
|
|
207
|
-
for (let i = 0; i < tokens.length; i++) {
|
|
208
|
-
for (let n = minWords; n <= maxWords; n++) {
|
|
209
|
-
extractNounEdgeGrams(n, tokens, i, nGrams, minWordLength, idx);
|
|
210
|
-
}
|
|
211
|
-
}
|
|
212
|
-
}
|
|
213
|
-
|
|
214
|
-
// \u2500\u2500 5. Score raw keyphrases \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
215
|
-
let keyphraseObjects: KeyphraseEntry[] = [];
|
|
216
|
-
for (let n = minWords; n <= maxWords; n++) {
|
|
217
|
-
if (!nGrams[n]) continue;
|
|
218
|
-
for (const [keyphrase, sentencesArr] of Object.entries(nGrams[n])) {
|
|
219
|
-
keyphraseObjects.push({
|
|
220
|
-
keyphrase,
|
|
221
|
-
sentences: sentencesArr,
|
|
222
|
-
words: n,
|
|
223
|
-
weight: sentencesArr.length * n,
|
|
224
|
-
});
|
|
225
|
-
}
|
|
226
|
-
}
|
|
227
|
-
|
|
228
|
-
// \u2500\u2500 6. Fold subphrases + deduplicate \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
229
|
-
const folded = foldSubphrases(keyphraseObjects);
|
|
230
|
-
const seen = new Set<string>();
|
|
231
|
-
keyphraseObjects = folded
|
|
232
|
-
.sort((a, b) => b.weight - a.weight)
|
|
233
|
-
.filter((k) => !seen.has(k.keyphrase) && seen.add(k.keyphrase))
|
|
234
|
-
.map((k) => ({
|
|
235
|
-
...k,
|
|
236
|
-
sentences: Array.isArray(k.sentences)
|
|
237
|
-
? [...new Set(k.sentences as number[])]
|
|
238
|
-
: k.sentences,
|
|
239
|
-
}));
|
|
240
|
-
|
|
241
|
-
// \u2500\u2500 7. IDF + wiki-entity + heavy-query weighting \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
242
|
-
keyphraseObjects = weightKeyphrasesBySpecificity(
|
|
243
|
-
keyphraseObjects,
|
|
244
|
-
phrasesModel,
|
|
245
|
-
heavyWeightQuery,
|
|
246
|
-
)
|
|
247
|
-
.filter((k) => k.keyphrase.length > minKeyPhraseLength)
|
|
248
|
-
.sort((a, b) => b.weight - a.weight);
|
|
249
|
-
|
|
250
|
-
// Fast path: return keyphrases without TextRank
|
|
251
|
-
if (optionSkipRanking) return keyphraseObjects;
|
|
252
|
-
|
|
253
|
-
// \u2500\u2500 8. Attach keyphrases to sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
254
|
-
for (const { keyphrase, sentences, weight } of keyphraseObjects) {
|
|
255
|
-
if (Array.isArray(sentences)) {
|
|
256
|
-
for (const si of sentences) {
|
|
257
|
-
sentenceKeysMap[si as number]?.keyphrases.push({ keyphrase, weight });
|
|
258
|
-
}
|
|
259
|
-
}
|
|
260
|
-
}
|
|
261
|
-
|
|
262
|
-
// \u2500\u2500 9. TextRank \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
263
|
-
const ranked = rankSentencesCentralToKeyphrase(sentenceKeysMap);
|
|
264
|
-
|
|
265
|
-
const topSentences = (ranked ?? [])
|
|
266
|
-
.sort((a, b) => b.weight - a.weight)
|
|
267
|
-
.slice(0, limitTopSentences)
|
|
268
|
-
.map((s) => ({
|
|
269
|
-
...s,
|
|
270
|
-
keyphrases: [...new Set(s.keyphrases.map((k) => k.keyphrase))],
|
|
271
|
-
}));
|
|
272
|
-
|
|
273
|
-
const keyphrases = keyphraseObjects.slice(0, limitTopKeyphrases).map((k) => ({
|
|
274
|
-
...k,
|
|
275
|
-
sentences: Array.isArray(k.sentences) ? k.sentences.join(",") : k.sentences,
|
|
276
|
-
}));
|
|
277
|
-
|
|
278
|
-
return { topSentences, keyphrases, sentences: sentencesArray };
|
|
279
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Main orchestrator for the SEEKTOPIC keyphrase extraction pipeline.
|
|
3
|
+
* Coordinates cleaning, segmentation, LLM extraction, and fallback n-gram ranking.
|
|
4
|
+
*/
|
|
5
|
+
import type {
|
|
6
|
+
NgramMap,
|
|
7
|
+
KeyphraseEntry,
|
|
8
|
+
SentenceEntry,
|
|
9
|
+
SEEKTOPICOptions,
|
|
10
|
+
SEEKTOPICResult,
|
|
11
|
+
} from "./types";
|
|
12
|
+
import { splitTextToSentences } from "../tokenize/text-to-sentences";
|
|
13
|
+
import { convertTextToTokens } from "../tokenize/text-to-topic-tokens";
|
|
14
|
+
import { extractNounEdgeGrams } from "./ngrams";
|
|
15
|
+
import { foldSubphrases } from "./fold-keyphrases";
|
|
16
|
+
import { weightKeyphrasesBySpecificity } from "./weight-keyphrases";
|
|
17
|
+
import { rankSentencesCentralToKeyphrase } from "./rank-sentences-keyphrases";
|
|
18
|
+
import { weighRelevanceConceptVectorMultiple } from "./vector-search";
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* ### SEEKTOPIC \u2014 Keyphrase & Sentence Extraction
|
|
22
|
+
*
|
|
23
|
+
* Pulls the most important phrases and sentences out of any document.
|
|
24
|
+
* Given raw text, it returns a ranked list of key concepts (e.g. "neural
|
|
25
|
+
* network", "climate change") and, optionally, the sentences that best
|
|
26
|
+
* summarise the document around those concepts.
|
|
27
|
+
*
|
|
28
|
+
* <img src="https://i.imgur.com/gZ4kI1V.png" width="360px" />
|
|
29
|
+
*
|
|
30
|
+
* **How it works \u2014 8-step pipeline:**
|
|
31
|
+
*
|
|
32
|
+
* 1. **Clean** \u2014 strip HTML tags and entities.
|
|
33
|
+
* 2. **Split** \u2014 break text into sentences, respecting abbreviations and URLs.
|
|
34
|
+
* 3. **Topic Extraction** *(LLM Path)* \u2014 sends the first 5000 words to an LLM
|
|
35
|
+
* to extract the most descriptive key topic phrases labeling the document.
|
|
36
|
+
* 4. **Vector Search** *(LLM Path)* \u2014 calculates cosine similarity embeddings
|
|
37
|
+
* for all sentences vs topic phrases, finding the top relevant sentences per topic.
|
|
38
|
+
* 5. **Tokenise & Extract** *(Fallback)* \u2014 if LLM fails, identify noun-anchored
|
|
39
|
+
* n-grams (1-N words).
|
|
40
|
+
* 6. **Score & Fold** *(Fallback)* \u2014 weight phrases and merge subsumed phrases.
|
|
41
|
+
* 7. **Re-weight** *(Fallback)* \u2014 boost Wikipedia entities and rare domain terms.
|
|
42
|
+
* 8. **TextRank** *(Fallback)* \u2014 build sentence similarity graph and run random walk.
|
|
43
|
+
*
|
|
44
|
+
* <video src="https://github.com/user-attachments/assets/73348d63-7671-4e20-8df9-29a13d5b0768"
|
|
45
|
+
* width="550px" controls />
|
|
46
|
+
*
|
|
47
|
+
* @param docText - Plain text or HTML document to analyse.
|
|
48
|
+
* @param options - Optional tuning parameters (word limits, thresholds, query bias).
|
|
49
|
+
* @returns
|
|
50
|
+
* - `optionSkipRanking: true` (default) \u2192 `KeyphraseEntry[]` sorted by weight.
|
|
51
|
+
* - `optionSkipRanking: false` \u2192 `SEEKTOPICResult` with `topSentences`,
|
|
52
|
+
* `keyphrases`, and the full `sentences` array.
|
|
53
|
+
*
|
|
54
|
+
* @example
|
|
55
|
+
* // Fast: just extract keyphrases
|
|
56
|
+
* const keyphrases = extractSEEKTOPIC(articleText, { phrasesModel });
|
|
57
|
+
* // \u2192 [{ keyphrase: "machine learning", weight: 84 }, ...]
|
|
58
|
+
*
|
|
59
|
+
* @example
|
|
60
|
+
* // Full: keyphrases + summary sentences, biased toward a search query
|
|
61
|
+
* const { topSentences, keyphrases } = extractSEEKTOPIC(articleText, {
|
|
62
|
+
* phrasesModel,
|
|
63
|
+
* optionSkipRanking: false,
|
|
64
|
+
* heavyWeightQuery: "transformer attention",
|
|
65
|
+
* limitTopSentences: 5,
|
|
66
|
+
* }) as SEEKTOPICResult;
|
|
67
|
+
*
|
|
68
|
+
* @author [ai-research-agent (2024)](https://airesearch.js.org)
|
|
69
|
+
* @category Topics
|
|
70
|
+
*/
|
|
71
|
+
export async function extractSEEKTOPIC(
|
|
72
|
+
docText: string,
|
|
73
|
+
options: SEEKTOPICOptions = {},
|
|
74
|
+
): Promise<KeyphraseEntry[] | SEEKTOPICResult> {
|
|
75
|
+
if (typeof docText !== "string") throw new Error("docText must be a string");
|
|
76
|
+
|
|
77
|
+
const {
|
|
78
|
+
phrasesModel,
|
|
79
|
+
maxWords = 2,
|
|
80
|
+
minWords = 1,
|
|
81
|
+
minWordLength = 3,
|
|
82
|
+
topKeyphrasesPercent = 0.5,
|
|
83
|
+
limitTopSentences = 5,
|
|
84
|
+
limitTopKeyphrases = 10,
|
|
85
|
+
minKeyPhraseLength = 5,
|
|
86
|
+
heavyWeightQuery = "",
|
|
87
|
+
removeHTML = true,
|
|
88
|
+
optionSkipRanking = true,
|
|
89
|
+
getEnv = () => "",
|
|
90
|
+
} = options;
|
|
91
|
+
|
|
92
|
+
// \u2500\u2500 1. Normalize HTML \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
93
|
+
let text = docText
|
|
94
|
+
.replace(/</g, " <")
|
|
95
|
+
.replace(/>/g, "> ")
|
|
96
|
+
.replace(/&.{2,5};/g, ""); // strip " & < >
|
|
97
|
+
if (removeHTML) text = text.replace(/<[^>]*>/g, "");
|
|
98
|
+
|
|
99
|
+
// \u2500\u2500 2. Sentence segmentation \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
100
|
+
const sentencesArray = splitTextToSentences(text);
|
|
101
|
+
|
|
102
|
+
// \u2500\u2500 3. LLM Topic Extraction (Starting Step) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
103
|
+
const first5kWords = text.split(/\s+/).slice(0, 5000).join(" ");
|
|
104
|
+
const llmPrompt = `Extract the 5-10 most important key topic phrases that best
|
|
105
|
+
label this document. Return exclusively a comma-separated list of the key topic
|
|
106
|
+
phrases. Document text: ${first5kWords}`;
|
|
107
|
+
|
|
108
|
+
let llmTopics: string[] = [];
|
|
109
|
+
try {
|
|
110
|
+
// `generate-language` moved into the chat-agent-toolkit workspace package.
|
|
111
|
+
// Load it lazily (and tolerate absence) since the LLM path is an optional
|
|
112
|
+
// enhancement — the n-gram fallback below covers extraction without it.
|
|
113
|
+
const { writeLanguageResponse } = await import("chat-agent-toolkit");
|
|
114
|
+
const aiResponse = await writeLanguageResponse({
|
|
115
|
+
query: llmPrompt,
|
|
116
|
+
provider: getEnv("LLM_PROVIDER") || "openai",
|
|
117
|
+
apiKey: getEnv("LLM_API_KEY") || getEnv("OPENAI_API_KEY") || "dummy",
|
|
118
|
+
});
|
|
119
|
+
if (aiResponse.content && !aiResponse.error) {
|
|
120
|
+
llmTopics = aiResponse.content
|
|
121
|
+
.split(",")
|
|
122
|
+
.map((s: string) => s.trim().replace(/^['"]|['"]$/g, ""))
|
|
123
|
+
.filter(Boolean);
|
|
124
|
+
}
|
|
125
|
+
} catch (e) {
|
|
126
|
+
console.error("LLM topic extraction failed", e);
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// \u2500\u2500 4. Use vector-search.ts for top topic sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
130
|
+
if (llmTopics.length > 0) {
|
|
131
|
+
if (optionSkipRanking) {
|
|
132
|
+
return llmTopics.map((topic: string) => ({
|
|
133
|
+
keyphrase: topic,
|
|
134
|
+
sentences: [],
|
|
135
|
+
words: topic.split(/\s+/).length,
|
|
136
|
+
weight: 100,
|
|
137
|
+
}));
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
const { limitTopSentences = 3 } = options;
|
|
141
|
+
const allRelevant = await weighRelevanceConceptVectorMultiple(
|
|
142
|
+
sentencesArray,
|
|
143
|
+
llmTopics,
|
|
144
|
+
options,
|
|
145
|
+
);
|
|
146
|
+
|
|
147
|
+
const keyphraseObjects: any[] = [];
|
|
148
|
+
const topSentences: any[] = [];
|
|
149
|
+
|
|
150
|
+
for (const topic of llmTopics) {
|
|
151
|
+
// @ts-ignore
|
|
152
|
+
const topicRelevant = allRelevant[topic] || [];
|
|
153
|
+
const topX = topicRelevant.slice(0, limitTopSentences);
|
|
154
|
+
|
|
155
|
+
const topicSentences = topX.map((r: any) => ({
|
|
156
|
+
text: r.content,
|
|
157
|
+
similarity: r.similarity,
|
|
158
|
+
}));
|
|
159
|
+
|
|
160
|
+
keyphraseObjects.push({
|
|
161
|
+
keyphrase: topic,
|
|
162
|
+
topSentences: topicSentences,
|
|
163
|
+
words: topic.split(/\s+/).length,
|
|
164
|
+
weight: 100,
|
|
165
|
+
});
|
|
166
|
+
|
|
167
|
+
for (const r of topX) {
|
|
168
|
+
const sIdx = sentencesArray.indexOf(r.content);
|
|
169
|
+
topSentences.push({
|
|
170
|
+
text: r.content,
|
|
171
|
+
index: sIdx !== -1 ? sIdx : 0,
|
|
172
|
+
keyphrases: [{ keyphrase: topic, weight: r.similarity }],
|
|
173
|
+
weight: r.similarity,
|
|
174
|
+
});
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
// Deduplicate topSentences
|
|
179
|
+
const uniqueSentences = Array.from(
|
|
180
|
+
new Map(topSentences.map((s: any) => [s.text, s])).values(),
|
|
181
|
+
);
|
|
182
|
+
|
|
183
|
+
return {
|
|
184
|
+
topSentences: uniqueSentences.sort(
|
|
185
|
+
(a: any, b: any) => b.weight - a.weight,
|
|
186
|
+
),
|
|
187
|
+
keyphrases: keyphraseObjects,
|
|
188
|
+
sentences: sentencesArray,
|
|
189
|
+
} as unknown as SEEKTOPICResult;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
// \u2500\u2500 FALLBACK: Tokenise + extract n-grams \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
193
|
+
const nGrams: NgramMap = {};
|
|
194
|
+
const sentenceKeysMap: SentenceEntry[] = [];
|
|
195
|
+
|
|
196
|
+
for (let idx = 0; idx < sentencesArray.length; idx++) {
|
|
197
|
+
const sentence = sentencesArray[idx];
|
|
198
|
+
const tokens = convertTextToTokens(sentence, { phrasesModel });
|
|
199
|
+
|
|
200
|
+
sentenceKeysMap.push({
|
|
201
|
+
text: sentence,
|
|
202
|
+
index: idx,
|
|
203
|
+
keyphrases: [],
|
|
204
|
+
weight: 0,
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
208
|
+
for (let n = minWords; n <= maxWords; n++) {
|
|
209
|
+
extractNounEdgeGrams(n, tokens, i, nGrams, minWordLength, idx);
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
// \u2500\u2500 5. Score raw keyphrases \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
215
|
+
let keyphraseObjects: KeyphraseEntry[] = [];
|
|
216
|
+
for (let n = minWords; n <= maxWords; n++) {
|
|
217
|
+
if (!nGrams[n]) continue;
|
|
218
|
+
for (const [keyphrase, sentencesArr] of Object.entries(nGrams[n])) {
|
|
219
|
+
keyphraseObjects.push({
|
|
220
|
+
keyphrase,
|
|
221
|
+
sentences: sentencesArr,
|
|
222
|
+
words: n,
|
|
223
|
+
weight: sentencesArr.length * n,
|
|
224
|
+
});
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
// \u2500\u2500 6. Fold subphrases + deduplicate \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
229
|
+
const folded = foldSubphrases(keyphraseObjects);
|
|
230
|
+
const seen = new Set<string>();
|
|
231
|
+
keyphraseObjects = folded
|
|
232
|
+
.sort((a, b) => b.weight - a.weight)
|
|
233
|
+
.filter((k) => !seen.has(k.keyphrase) && seen.add(k.keyphrase))
|
|
234
|
+
.map((k) => ({
|
|
235
|
+
...k,
|
|
236
|
+
sentences: Array.isArray(k.sentences)
|
|
237
|
+
? [...new Set(k.sentences as number[])]
|
|
238
|
+
: k.sentences,
|
|
239
|
+
}));
|
|
240
|
+
|
|
241
|
+
// \u2500\u2500 7. IDF + wiki-entity + heavy-query weighting \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
242
|
+
keyphraseObjects = weightKeyphrasesBySpecificity(
|
|
243
|
+
keyphraseObjects,
|
|
244
|
+
phrasesModel,
|
|
245
|
+
heavyWeightQuery,
|
|
246
|
+
)
|
|
247
|
+
.filter((k) => k.keyphrase.length > minKeyPhraseLength)
|
|
248
|
+
.sort((a, b) => b.weight - a.weight);
|
|
249
|
+
|
|
250
|
+
// Fast path: return keyphrases without TextRank
|
|
251
|
+
if (optionSkipRanking) return keyphraseObjects;
|
|
252
|
+
|
|
253
|
+
// \u2500\u2500 8. Attach keyphrases to sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
254
|
+
for (const { keyphrase, sentences, weight } of keyphraseObjects) {
|
|
255
|
+
if (Array.isArray(sentences)) {
|
|
256
|
+
for (const si of sentences) {
|
|
257
|
+
sentenceKeysMap[si as number]?.keyphrases.push({ keyphrase, weight });
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
// \u2500\u2500 9. TextRank \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
263
|
+
const ranked = rankSentencesCentralToKeyphrase(sentenceKeysMap);
|
|
264
|
+
|
|
265
|
+
const topSentences = (ranked ?? [])
|
|
266
|
+
.sort((a, b) => b.weight - a.weight)
|
|
267
|
+
.slice(0, limitTopSentences)
|
|
268
|
+
.map((s) => ({
|
|
269
|
+
...s,
|
|
270
|
+
keyphrases: [...new Set(s.keyphrases.map((k) => k.keyphrase))],
|
|
271
|
+
}));
|
|
272
|
+
|
|
273
|
+
const keyphrases = keyphraseObjects.slice(0, limitTopKeyphrases).map((k) => ({
|
|
274
|
+
...k,
|
|
275
|
+
sentences: Array.isArray(k.sentences) ? k.sentences.join(",") : k.sentences,
|
|
276
|
+
}));
|
|
277
|
+
|
|
278
|
+
return { topSentences, keyphrases, sentences: sentencesArray };
|
|
279
|
+
}
|