extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Main orchestrator for the SEEKTOPIC keyphrase extraction pipeline.
|
|
3
|
+
* Coordinates cleaning, segmentation, LLM extraction, and fallback n-gram ranking.
|
|
4
|
+
*/
|
|
5
|
+
import type {
|
|
6
|
+
NgramMap,
|
|
7
|
+
KeyphraseEntry,
|
|
8
|
+
SentenceEntry,
|
|
9
|
+
SEEKTOPICOptions,
|
|
10
|
+
SEEKTOPICResult,
|
|
11
|
+
} from "./types";
|
|
12
|
+
import { splitTextToSentences } from "../tokenize/text-to-sentences";
|
|
13
|
+
import { convertTextToTokens } from "../tokenize/text-to-topic-tokens";
|
|
14
|
+
import { extractNounEdgeGrams } from "./ngrams";
|
|
15
|
+
import { foldSubphrases } from "./fold-keyphrases";
|
|
16
|
+
import { weightKeyphrasesBySpecificity } from "./weight-keyphrases";
|
|
17
|
+
import { rankSentencesCentralToKeyphrase } from "./rank-sentences-keyphrases";
|
|
18
|
+
import { weighRelevanceConceptVectorMultiple } from "./vector-search";
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* ### SEEKTOPIC \u2014 Keyphrase & Sentence Extraction
|
|
22
|
+
*
|
|
23
|
+
* Pulls the most important phrases and sentences out of any document.
|
|
24
|
+
* Given raw text, it returns a ranked list of key concepts (e.g. "neural
|
|
25
|
+
* network", "climate change") and, optionally, the sentences that best
|
|
26
|
+
* summarise the document around those concepts.
|
|
27
|
+
*
|
|
28
|
+
* <img src="https://i.imgur.com/gZ4kI1V.png" width="360px" />
|
|
29
|
+
*
|
|
30
|
+
* **How it works \u2014 8-step pipeline:**
|
|
31
|
+
*
|
|
32
|
+
* 1. **Clean** \u2014 strip HTML tags and entities.
|
|
33
|
+
* 2. **Split** \u2014 break text into sentences, respecting abbreviations and URLs.
|
|
34
|
+
* 3. **Topic Extraction** *(LLM Path)* \u2014 sends the first 5000 words to an LLM
|
|
35
|
+
* to extract the most descriptive key topic phrases labeling the document.
|
|
36
|
+
* 4. **Vector Search** *(LLM Path)* \u2014 calculates cosine similarity embeddings
|
|
37
|
+
* for all sentences vs topic phrases, finding the top relevant sentences per topic.
|
|
38
|
+
* 5. **Tokenise & Extract** *(Fallback)* \u2014 if LLM fails, identify noun-anchored
|
|
39
|
+
* n-grams (1-N words).
|
|
40
|
+
* 6. **Score & Fold** *(Fallback)* \u2014 weight phrases and merge subsumed phrases.
|
|
41
|
+
* 7. **Re-weight** *(Fallback)* \u2014 boost Wikipedia entities and rare domain terms.
|
|
42
|
+
* 8. **TextRank** *(Fallback)* \u2014 build sentence similarity graph and run random walk.
|
|
43
|
+
*
|
|
44
|
+
* <video src="https://github.com/user-attachments/assets/73348d63-7671-4e20-8df9-29a13d5b0768"
|
|
45
|
+
* width="550px" controls />
|
|
46
|
+
*
|
|
47
|
+
* @param docText - Plain text or HTML document to analyse.
|
|
48
|
+
* @param options - Optional tuning parameters (word limits, thresholds, query bias).
|
|
49
|
+
* @returns
|
|
50
|
+
* - `optionSkipRanking: true` (default) \u2192 `KeyphraseEntry[]` sorted by weight.
|
|
51
|
+
* - `optionSkipRanking: false` \u2192 `SEEKTOPICResult` with `topSentences`,
|
|
52
|
+
* `keyphrases`, and the full `sentences` array.
|
|
53
|
+
*
|
|
54
|
+
* @example
|
|
55
|
+
* // Fast: just extract keyphrases
|
|
56
|
+
* const keyphrases = extractSEEKTOPIC(articleText, { phrasesModel });
|
|
57
|
+
* // \u2192 [{ keyphrase: "machine learning", weight: 84 }, ...]
|
|
58
|
+
*
|
|
59
|
+
* @example
|
|
60
|
+
* // Full: keyphrases + summary sentences, biased toward a search query
|
|
61
|
+
* const { topSentences, keyphrases } = extractSEEKTOPIC(articleText, {
|
|
62
|
+
* phrasesModel,
|
|
63
|
+
* optionSkipRanking: false,
|
|
64
|
+
* heavyWeightQuery: "transformer attention",
|
|
65
|
+
* limitTopSentences: 5,
|
|
66
|
+
* }) as SEEKTOPICResult;
|
|
67
|
+
*
|
|
68
|
+
* @author [ai-research-agent (2024)](https://airesearch.js.org)
|
|
69
|
+
* @category Topics
|
|
70
|
+
*/
|
|
71
|
+
export async function extractSEEKTOPIC(
|
|
72
|
+
docText: string,
|
|
73
|
+
options: SEEKTOPICOptions = {},
|
|
74
|
+
): Promise<KeyphraseEntry[] | SEEKTOPICResult> {
|
|
75
|
+
if (typeof docText !== "string") throw new Error("docText must be a string");
|
|
76
|
+
|
|
77
|
+
const {
|
|
78
|
+
phrasesModel,
|
|
79
|
+
maxWords = 2,
|
|
80
|
+
minWords = 1,
|
|
81
|
+
minWordLength = 3,
|
|
82
|
+
topKeyphrasesPercent = 0.5,
|
|
83
|
+
limitTopSentences = 5,
|
|
84
|
+
limitTopKeyphrases = 10,
|
|
85
|
+
minKeyPhraseLength = 5,
|
|
86
|
+
heavyWeightQuery = "",
|
|
87
|
+
removeHTML = true,
|
|
88
|
+
optionSkipRanking = true,
|
|
89
|
+
getEnv = () => "",
|
|
90
|
+
} = options;
|
|
91
|
+
|
|
92
|
+
// \u2500\u2500 1. Normalize HTML \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
93
|
+
let text = docText
|
|
94
|
+
.replace(/</g, " <")
|
|
95
|
+
.replace(/>/g, "> ")
|
|
96
|
+
.replace(/&.{2,5};/g, ""); // strip " & < >
|
|
97
|
+
if (removeHTML) text = text.replace(/<[^>]*>/g, "");
|
|
98
|
+
|
|
99
|
+
// \u2500\u2500 2. Sentence segmentation \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
100
|
+
const sentencesArray = splitTextToSentences(text);
|
|
101
|
+
|
|
102
|
+
// \u2500\u2500 3. LLM Topic Extraction (Starting Step) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
103
|
+
const first5kWords = text.split(/\s+/).slice(0, 5000).join(" ");
|
|
104
|
+
const llmPrompt = `Extract the 5-10 most important key topic phrases that best
|
|
105
|
+
label this document. Return exclusively a comma-separated list of the key topic
|
|
106
|
+
phrases. Document text: ${first5kWords}`;
|
|
107
|
+
|
|
108
|
+
let llmTopics: string[] = [];
|
|
109
|
+
try {
|
|
110
|
+
// `generate-language` moved into the chat-agent-toolkit workspace package.
|
|
111
|
+
// Load it lazily (and tolerate absence) since the LLM path is an optional
|
|
112
|
+
// enhancement — the n-gram fallback below covers extraction without it.
|
|
113
|
+
const { writeLanguageResponse } = await import("chat-agent-toolkit");
|
|
114
|
+
const aiResponse = await writeLanguageResponse({
|
|
115
|
+
query: llmPrompt,
|
|
116
|
+
provider: getEnv("LLM_PROVIDER") || "openai",
|
|
117
|
+
apiKey: getEnv("LLM_API_KEY") || getEnv("OPENAI_API_KEY") || "dummy",
|
|
118
|
+
});
|
|
119
|
+
if (aiResponse.content && !aiResponse.error) {
|
|
120
|
+
llmTopics = aiResponse.content
|
|
121
|
+
.split(",")
|
|
122
|
+
.map((s: string) => s.trim().replace(/^['"]|['"]$/g, ""))
|
|
123
|
+
.filter(Boolean);
|
|
124
|
+
}
|
|
125
|
+
} catch (e) {
|
|
126
|
+
console.error("LLM topic extraction failed", e);
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// \u2500\u2500 4. Use vector-search.ts for top topic sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
130
|
+
if (llmTopics.length > 0) {
|
|
131
|
+
if (optionSkipRanking) {
|
|
132
|
+
return llmTopics.map((topic: string) => ({
|
|
133
|
+
keyphrase: topic,
|
|
134
|
+
sentences: [],
|
|
135
|
+
words: topic.split(/\s+/).length,
|
|
136
|
+
weight: 100,
|
|
137
|
+
}));
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
const { limitTopSentences = 3 } = options;
|
|
141
|
+
const allRelevant = await weighRelevanceConceptVectorMultiple(
|
|
142
|
+
sentencesArray,
|
|
143
|
+
llmTopics,
|
|
144
|
+
options,
|
|
145
|
+
);
|
|
146
|
+
|
|
147
|
+
const keyphraseObjects: any[] = [];
|
|
148
|
+
const topSentences: any[] = [];
|
|
149
|
+
|
|
150
|
+
for (const topic of llmTopics) {
|
|
151
|
+
// @ts-ignore
|
|
152
|
+
const topicRelevant = allRelevant[topic] || [];
|
|
153
|
+
const topX = topicRelevant.slice(0, limitTopSentences);
|
|
154
|
+
|
|
155
|
+
const topicSentences = topX.map((r: any) => ({
|
|
156
|
+
text: r.content,
|
|
157
|
+
similarity: r.similarity,
|
|
158
|
+
}));
|
|
159
|
+
|
|
160
|
+
keyphraseObjects.push({
|
|
161
|
+
keyphrase: topic,
|
|
162
|
+
topSentences: topicSentences,
|
|
163
|
+
words: topic.split(/\s+/).length,
|
|
164
|
+
weight: 100,
|
|
165
|
+
});
|
|
166
|
+
|
|
167
|
+
for (const r of topX) {
|
|
168
|
+
const sIdx = sentencesArray.indexOf(r.content);
|
|
169
|
+
topSentences.push({
|
|
170
|
+
text: r.content,
|
|
171
|
+
index: sIdx !== -1 ? sIdx : 0,
|
|
172
|
+
keyphrases: [{ keyphrase: topic, weight: r.similarity }],
|
|
173
|
+
weight: r.similarity,
|
|
174
|
+
});
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
// Deduplicate topSentences
|
|
179
|
+
const uniqueSentences = Array.from(
|
|
180
|
+
new Map(topSentences.map((s: any) => [s.text, s])).values(),
|
|
181
|
+
);
|
|
182
|
+
|
|
183
|
+
return {
|
|
184
|
+
topSentences: uniqueSentences.sort(
|
|
185
|
+
(a: any, b: any) => b.weight - a.weight,
|
|
186
|
+
),
|
|
187
|
+
keyphrases: keyphraseObjects,
|
|
188
|
+
sentences: sentencesArray,
|
|
189
|
+
} as unknown as SEEKTOPICResult;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
// \u2500\u2500 FALLBACK: Tokenise + extract n-grams \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
193
|
+
const nGrams: NgramMap = {};
|
|
194
|
+
const sentenceKeysMap: SentenceEntry[] = [];
|
|
195
|
+
|
|
196
|
+
for (let idx = 0; idx < sentencesArray.length; idx++) {
|
|
197
|
+
const sentence = sentencesArray[idx];
|
|
198
|
+
const tokens = convertTextToTokens(sentence, { phrasesModel });
|
|
199
|
+
|
|
200
|
+
sentenceKeysMap.push({
|
|
201
|
+
text: sentence,
|
|
202
|
+
index: idx,
|
|
203
|
+
keyphrases: [],
|
|
204
|
+
weight: 0,
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
208
|
+
for (let n = minWords; n <= maxWords; n++) {
|
|
209
|
+
extractNounEdgeGrams(n, tokens, i, nGrams, minWordLength, idx);
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
// \u2500\u2500 5. Score raw keyphrases \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
215
|
+
let keyphraseObjects: KeyphraseEntry[] = [];
|
|
216
|
+
for (let n = minWords; n <= maxWords; n++) {
|
|
217
|
+
if (!nGrams[n]) continue;
|
|
218
|
+
for (const [keyphrase, sentencesArr] of Object.entries(nGrams[n])) {
|
|
219
|
+
keyphraseObjects.push({
|
|
220
|
+
keyphrase,
|
|
221
|
+
sentences: sentencesArr,
|
|
222
|
+
words: n,
|
|
223
|
+
weight: sentencesArr.length * n,
|
|
224
|
+
});
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
// \u2500\u2500 6. Fold subphrases + deduplicate \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
229
|
+
const folded = foldSubphrases(keyphraseObjects);
|
|
230
|
+
const seen = new Set<string>();
|
|
231
|
+
keyphraseObjects = folded
|
|
232
|
+
.sort((a, b) => b.weight - a.weight)
|
|
233
|
+
.filter((k) => !seen.has(k.keyphrase) && seen.add(k.keyphrase))
|
|
234
|
+
.map((k) => ({
|
|
235
|
+
...k,
|
|
236
|
+
sentences: Array.isArray(k.sentences)
|
|
237
|
+
? [...new Set(k.sentences as number[])]
|
|
238
|
+
: k.sentences,
|
|
239
|
+
}));
|
|
240
|
+
|
|
241
|
+
// \u2500\u2500 7. IDF + wiki-entity + heavy-query weighting \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
242
|
+
keyphraseObjects = weightKeyphrasesBySpecificity(
|
|
243
|
+
keyphraseObjects,
|
|
244
|
+
phrasesModel,
|
|
245
|
+
heavyWeightQuery,
|
|
246
|
+
)
|
|
247
|
+
.filter((k) => k.keyphrase.length > minKeyPhraseLength)
|
|
248
|
+
.sort((a, b) => b.weight - a.weight);
|
|
249
|
+
|
|
250
|
+
// Fast path: return keyphrases without TextRank
|
|
251
|
+
if (optionSkipRanking) return keyphraseObjects;
|
|
252
|
+
|
|
253
|
+
// \u2500\u2500 8. Attach keyphrases to sentences \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
254
|
+
for (const { keyphrase, sentences, weight } of keyphraseObjects) {
|
|
255
|
+
if (Array.isArray(sentences)) {
|
|
256
|
+
for (const si of sentences) {
|
|
257
|
+
sentenceKeysMap[si as number]?.keyphrases.push({ keyphrase, weight });
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
// \u2500\u2500 9. TextRank \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
|
|
263
|
+
const ranked = rankSentencesCentralToKeyphrase(sentenceKeysMap);
|
|
264
|
+
|
|
265
|
+
const topSentences = (ranked ?? [])
|
|
266
|
+
.sort((a, b) => b.weight - a.weight)
|
|
267
|
+
.slice(0, limitTopSentences)
|
|
268
|
+
.map((s) => ({
|
|
269
|
+
...s,
|
|
270
|
+
keyphrases: [...new Set(s.keyphrases.map((k) => k.keyphrase))],
|
|
271
|
+
}));
|
|
272
|
+
|
|
273
|
+
const keyphrases = keyphraseObjects.slice(0, limitTopKeyphrases).map((k) => ({
|
|
274
|
+
...k,
|
|
275
|
+
sentences: Array.isArray(k.sentences) ? k.sentences.join(",") : k.sentences,
|
|
276
|
+
}));
|
|
277
|
+
|
|
278
|
+
return { topSentences, keyphrases, sentences: sentencesArray };
|
|
279
|
+
}
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Type definitions for the SEEKTOPIC engine and ranking utilities.
|
|
3
|
+
*/
|
|
4
|
+
import type {
|
|
5
|
+
TopicToken,
|
|
6
|
+
PhrasesModel,
|
|
7
|
+
} from "../tokenize/text-to-topic-tokens";
|
|
8
|
+
|
|
9
|
+
export type { TopicToken, PhrasesModel };
|
|
10
|
+
|
|
11
|
+
/** Map from n-gram size \u2192 { phrase text \u2192 sentence indices } */
|
|
12
|
+
export type NgramMap = Record<number, Record<string, number[]>>;
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* A ranked keyphrase with its scoring metadata and the sentence indices it appears in.
|
|
16
|
+
*/
|
|
17
|
+
export interface KeyphraseEntry {
|
|
18
|
+
/** The keyphrase text */
|
|
19
|
+
keyphrase: string;
|
|
20
|
+
/** Indices of sentences containing this keyphrase (or comma-separated string form) */
|
|
21
|
+
sentences: number[] | string;
|
|
22
|
+
/** Sentences representing this keyphrase directly with their relevance similarity */
|
|
23
|
+
topSentences?: Array<{ text: string; similarity: number }>;
|
|
24
|
+
/** Number of words in the keyphrase */
|
|
25
|
+
words: number;
|
|
26
|
+
/** Composite ranking weight */
|
|
27
|
+
weight: number;
|
|
28
|
+
/** True if the phrase is a Wikipedia-linked entity */
|
|
29
|
+
wiki?: boolean;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* A sentence node used during graph construction and TextRank scoring.
|
|
34
|
+
*/
|
|
35
|
+
export interface SentenceEntry {
|
|
36
|
+
text: string;
|
|
37
|
+
index: number;
|
|
38
|
+
keyphrases: Array<{ keyphrase: string; weight: number }>;
|
|
39
|
+
weight: number;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* A sentence in the final SEEKTOPIC output \u2014 keyphrases are resolved to strings.
|
|
44
|
+
*/
|
|
45
|
+
export interface SentenceResult {
|
|
46
|
+
text: string;
|
|
47
|
+
index: number;
|
|
48
|
+
keyphrases: string[];
|
|
49
|
+
weight: number;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Options for {@link extractSEEKTOPIC} */
|
|
53
|
+
export interface SEEKTOPICOptions {
|
|
54
|
+
/** Trie phrases model for wiki-phrase tokenization */
|
|
55
|
+
phrasesModel?: PhrasesModel;
|
|
56
|
+
/** Maximum words per keyphrase (default 2) */
|
|
57
|
+
maxWords?: number;
|
|
58
|
+
/** Minimum words per keyphrase (default 1) */
|
|
59
|
+
minWords?: number;
|
|
60
|
+
/** Minimum character length of any word in a keyphrase (default 3) */
|
|
61
|
+
minWordLength?: number;
|
|
62
|
+
/** Fraction of all keyphrases to use as the TextRank graph limit (default 0.5) */
|
|
63
|
+
topKeyphrasesPercent?: number;
|
|
64
|
+
/** Max sentences to return in full-ranking mode (default 5) */
|
|
65
|
+
limitTopSentences?: number;
|
|
66
|
+
/** Max keyphrases to return (default 10) */
|
|
67
|
+
limitTopKeyphrases?: number;
|
|
68
|
+
/** Minimum character length of the full keyphrase string (default 5) */
|
|
69
|
+
minKeyPhraseLength?: number;
|
|
70
|
+
/** Query string to bias keyphrase weights toward (default "") */
|
|
71
|
+
heavyWeightQuery?: string;
|
|
72
|
+
/** Strip HTML tags before processing (default true) */
|
|
73
|
+
removeHTML?: boolean;
|
|
74
|
+
/** Return only keyphrases without running TextRank (default true) */
|
|
75
|
+
optionSkipRanking?: boolean;
|
|
76
|
+
getEnv?: (key: string) => string | undefined;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** Full SEEKTOPIC output when `optionSkipRanking` is false */
|
|
80
|
+
export interface SEEKTOPICResult {
|
|
81
|
+
topSentences: SentenceResult[];
|
|
82
|
+
keyphrases: Array<Omit<KeyphraseEntry, "sentences"> & { sentences: string }>;
|
|
83
|
+
sentences: string[];
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Options for {@link rankSentencesCentralToKeyphrase} */
|
|
87
|
+
export interface RankOptions {
|
|
88
|
+
/** Number of random-walk steps (default 1000) */
|
|
89
|
+
iterations?: number;
|
|
90
|
+
/** Steps between forced random resets to avoid cluster traps (default 100) */
|
|
91
|
+
resetInterval?: number;
|
|
92
|
+
}
|
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
import grab from "../utils/grab";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Text embeddings convert words or phrases into numerical vectors in a high-dimensional
|
|
5
|
+
* space, where each dimension represents a semantic feature extracted by a model like
|
|
6
|
+
* MiniLM-L6-v2. In this concept space, words with similar meanings have vectors that
|
|
7
|
+
* are close together, allowing for quantitative comparisons of semantic similarity.
|
|
8
|
+
* These vector representations enable powerful applications in natural language processing,
|
|
9
|
+
* including semantic search, text classification, and clustering, by leveraging the
|
|
10
|
+
* geometric properties of the embedding space to capture and analyze the relationships
|
|
11
|
+
* between words and concepts.
|
|
12
|
+
* [Text Embeddings, Classification, and Semantic Search
|
|
13
|
+
* (Youtube)](https://www.youtube.com/watch?v=sNa_uiqSlJo&t=129s)
|
|
14
|
+
*
|
|
15
|
+
* <img src="https://i.imgur.com/wtJqEqX.png" width="350" />
|
|
16
|
+
* @param {string} text - The text to embed.
|
|
17
|
+
* @param {Object} [options]
|
|
18
|
+
* @param {AutoTokenizer} options.pipeline
|
|
19
|
+
* - The pipeline to use for embedding.
|
|
20
|
+
* @param {number} options.precision default=4 - The number of decimal places to round to.
|
|
21
|
+
* @returns {Promise<{embeddingsDict: Object.<string, number[]>, embedding: number[]}>}
|
|
22
|
+
* @category Similarity
|
|
23
|
+
*/
|
|
24
|
+
export async function convertTextToEmbedding(
|
|
25
|
+
text: string | string[],
|
|
26
|
+
options: any = {},
|
|
27
|
+
) {
|
|
28
|
+
var { precision = 4, pipeline } = options;
|
|
29
|
+
|
|
30
|
+
if (!pipeline) pipeline = await getEmbeddingModel();
|
|
31
|
+
|
|
32
|
+
const embedding = await pipeline(text, { pooling: "mean", normalize: true });
|
|
33
|
+
|
|
34
|
+
// Check if input was an array to determine output format
|
|
35
|
+
if (Array.isArray(text)) {
|
|
36
|
+
return embedding
|
|
37
|
+
.tolist()
|
|
38
|
+
.map((vec) => vec.map((num) => parseFloat(num.toFixed(precision))));
|
|
39
|
+
} else {
|
|
40
|
+
const roundedEmbedding = Array.from((embedding as any).data).map(
|
|
41
|
+
(num: any) => parseFloat((num as any).toFixed(precision)),
|
|
42
|
+
);
|
|
43
|
+
return roundedEmbedding;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Calculate the semantic similarity between one text and a list of
|
|
49
|
+
* other sentences by comparing their embeddings.
|
|
50
|
+
* https://huggingface.co/docs/api-inference/detailed_parameters#sentence-similarity-task
|
|
51
|
+
*
|
|
52
|
+
* <img src="https://i.imgur.com/ex2UWnu.png" width="350px" />
|
|
53
|
+
* @param {string} source_sentence The string that you wish to
|
|
54
|
+
* compare the other strings with. This can be a phrase, sentence,
|
|
55
|
+
* or longer passage, depending on the model being used.
|
|
56
|
+
* @param {Array<string>} sentences A list of strings which will be compared
|
|
57
|
+
* against the source_sentence.
|
|
58
|
+
* @param {Object} [options]
|
|
59
|
+
* @param {string} options.model default="sentence-transformers/all-MiniLM-L6-v2"
|
|
60
|
+
* @param {string} options.HF_API_KEY Required https://huggingface.co/settings/tokens
|
|
61
|
+
* @returns array of 0-1 similarity scores for each sentence
|
|
62
|
+
* @category Similarity
|
|
63
|
+
*/
|
|
64
|
+
export async function weighRelevanceConceptVectorAPI(
|
|
65
|
+
source_sentence,
|
|
66
|
+
sentences,
|
|
67
|
+
options = {},
|
|
68
|
+
) {
|
|
69
|
+
var { model = "sentence-transformers/all-MiniLM-L6-v2", HF_API_KEY = "" } =
|
|
70
|
+
options as any;
|
|
71
|
+
|
|
72
|
+
if (!HF_API_KEY) return { error: "No API key" };
|
|
73
|
+
|
|
74
|
+
const url = `https://api-inference.huggingface.co/models/${model}`;
|
|
75
|
+
try {
|
|
76
|
+
return await grab(url, {
|
|
77
|
+
method: "POST",
|
|
78
|
+
headers: {
|
|
79
|
+
Authorization: `Bearer ${HF_API_KEY}`,
|
|
80
|
+
},
|
|
81
|
+
body: JSON.stringify({
|
|
82
|
+
inputs: {
|
|
83
|
+
source_sentence,
|
|
84
|
+
sentences,
|
|
85
|
+
},
|
|
86
|
+
}),
|
|
87
|
+
});
|
|
88
|
+
} catch (error) {
|
|
89
|
+
console.error("API request failed:", error);
|
|
90
|
+
return null;
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Initialize HuggingFace Transformers pipeline for embedding text.
|
|
96
|
+
*
|
|
97
|
+
* <img src="https://i.imgur.com/3R5Tsrf.png" width="350px" />
|
|
98
|
+
* @param {Object} [options]
|
|
99
|
+
* @param {string} options.pipelineName default "feature-extraction",
|
|
100
|
+
* @param {string} options.modelName default="Xenova/all-MiniLM-L6-v2" -
|
|
101
|
+
* The name of the model to use
|
|
102
|
+
* @returns {Promise<import("@huggingface/transformers").AutoTokenizer>} The pipeline.
|
|
103
|
+
* @category Similarity
|
|
104
|
+
*/
|
|
105
|
+
export async function getEmbeddingModel(options: any = {}) {
|
|
106
|
+
const { pipeline } = await (import("@huggingface/transformers") as Promise<any>);
|
|
107
|
+
const {
|
|
108
|
+
pipelineName = "feature-extraction",
|
|
109
|
+
modelName = "Xenova/all-MiniLM-L6-v2",
|
|
110
|
+
quantized = true,
|
|
111
|
+
gpu = false,
|
|
112
|
+
} = options;
|
|
113
|
+
|
|
114
|
+
return await pipeline(pipelineName, modelName, {
|
|
115
|
+
quantized,
|
|
116
|
+
dtype: "fp32",
|
|
117
|
+
device: gpu ? "webgpu" : "cpu",
|
|
118
|
+
} as any);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* [Cosine similarity](https://en.wikipedia.org/wiki/Cosine_similarity) gets similarity of two
|
|
123
|
+
* vectors by whether they have the same direction (similar) or are poles apart. Cosine similarity
|
|
124
|
+
* is often used with text representations to compare how similar two documents or sentences
|
|
125
|
+
* are to each other. The output of cosine similarity ranges from -1 to 1, where -1 means the
|
|
126
|
+
* two vectors are completely dissimilar, and 1 indicates maximum similarity.
|
|
127
|
+
* @param {Array<number>} vectorA
|
|
128
|
+
* @param {Array<number>} vectorB
|
|
129
|
+
* @returns {number} -1 to 1 similarity score
|
|
130
|
+
*/
|
|
131
|
+
export function calculateCosineSimilarity(vectorA, vectorB) {
|
|
132
|
+
return (
|
|
133
|
+
vectorA.reduce((sum, a, i) => sum + a * vectorB[i], 0) /
|
|
134
|
+
(Math.sqrt(vectorA.reduce((sum, a) => sum + a * a, 0)) *
|
|
135
|
+
Math.sqrt(vectorB.reduce((sum, b) => sum + b * b, 0)))
|
|
136
|
+
);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
///OLDER ======================
|
|
140
|
+
/**
|
|
141
|
+
* Rerank documents's chunks based on relevance to query,
|
|
142
|
+
* based on cosine similarity of their concept vectors generated
|
|
143
|
+
* by a 20MB MiniLM transformer model downloaded locally.
|
|
144
|
+
*
|
|
145
|
+
* [A Complete Overview of Word Embeddings](https://www.youtube.com/watch?v=5MaWmXwxFNQ&t=323s)
|
|
146
|
+
* @param {Array<string>} documents
|
|
147
|
+
* @param {string} query
|
|
148
|
+
* @param {Object} [options]
|
|
149
|
+
* @returns {Promise<Array<{content: string, similarity: number}>>}
|
|
150
|
+
* @category Similarity
|
|
151
|
+
*/
|
|
152
|
+
export async function weighRelevanceConceptVector(
|
|
153
|
+
documents,
|
|
154
|
+
query,
|
|
155
|
+
options = {},
|
|
156
|
+
) {
|
|
157
|
+
const docEmbeddings = await convertTextToEmbedding(documents, options);
|
|
158
|
+
|
|
159
|
+
const queryEmbedding = await convertTextToEmbedding(query, options);
|
|
160
|
+
|
|
161
|
+
let sortedDocs = docEmbeddings
|
|
162
|
+
.map((docEmbedding, i) => ({
|
|
163
|
+
index: i,
|
|
164
|
+
similarity: calculateCosineSimilarity(queryEmbedding, docEmbedding),
|
|
165
|
+
}))
|
|
166
|
+
.sort((a, b) => b.similarity - a.similarity)
|
|
167
|
+
.map(({ index, similarity }) => ({
|
|
168
|
+
content: documents[index],
|
|
169
|
+
similarity,
|
|
170
|
+
}));
|
|
171
|
+
|
|
172
|
+
return sortedDocs;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/**
|
|
176
|
+
* Rerank documents's chunks based on relevance to multiple queries,
|
|
177
|
+
* optimizing by embedding documents only once.
|
|
178
|
+
*
|
|
179
|
+
* @param {Array<string>} documents
|
|
180
|
+
* @param {Array<string>} queries
|
|
181
|
+
* @param {Object} [options]
|
|
182
|
+
* @returns {Promise<Object<string, Array<{content: string, similarity: number}>>>}
|
|
183
|
+
* @category Similarity
|
|
184
|
+
*/
|
|
185
|
+
export async function weighRelevanceConceptVectorMultiple(
|
|
186
|
+
documents,
|
|
187
|
+
queries,
|
|
188
|
+
options = {},
|
|
189
|
+
) {
|
|
190
|
+
if (!documents || documents.length === 0 || !queries || queries.length === 0)
|
|
191
|
+
return {};
|
|
192
|
+
|
|
193
|
+
const pipeline = (options as any).pipeline;
|
|
194
|
+
const embedder = pipeline || (await getEmbeddingModel(options));
|
|
195
|
+
|
|
196
|
+
// Embed documents once
|
|
197
|
+
const docEmbeddings = await convertTextToEmbedding(documents, {
|
|
198
|
+
...options,
|
|
199
|
+
pipeline: embedder,
|
|
200
|
+
});
|
|
201
|
+
// Embed all queries at once
|
|
202
|
+
const queryEmbeddings = await convertTextToEmbedding(queries, {
|
|
203
|
+
...options,
|
|
204
|
+
pipeline: embedder,
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
// Ensure queryEmbeddings is a 2D array even if queries was a single string wrapped in an array
|
|
208
|
+
const qEmbeds = Array.isArray(queryEmbeddings[0])
|
|
209
|
+
? queryEmbeddings
|
|
210
|
+
: [queryEmbeddings];
|
|
211
|
+
|
|
212
|
+
const resultsByQuery = {};
|
|
213
|
+
|
|
214
|
+
queries.forEach((query, qIdx) => {
|
|
215
|
+
const queryEmbedding = qEmbeds[qIdx];
|
|
216
|
+
|
|
217
|
+
let sortedDocs = docEmbeddings
|
|
218
|
+
.map((docEmbedding, i) => ({
|
|
219
|
+
index: i,
|
|
220
|
+
similarity: calculateCosineSimilarity(queryEmbedding, docEmbedding),
|
|
221
|
+
}))
|
|
222
|
+
.sort((a, b) => b.similarity - a.similarity)
|
|
223
|
+
.map(({ index, similarity }) => ({
|
|
224
|
+
content: documents[index],
|
|
225
|
+
similarity,
|
|
226
|
+
}));
|
|
227
|
+
|
|
228
|
+
resultsByQuery[query] = sortedDocs;
|
|
229
|
+
});
|
|
230
|
+
|
|
231
|
+
return resultsByQuery;
|
|
232
|
+
}
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Scoring logic for keyphrases based on Wiki-entities, specificity, and query bias.
|
|
3
|
+
* Implements the ranking refinement steps of the SEEKTOPIC pipeline.
|
|
4
|
+
*/
|
|
5
|
+
import type { KeyphraseEntry, PhrasesModel } from "./types";
|
|
6
|
+
import { convertTextToTokens } from "../tokenize/text-to-topic-tokens";
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Applies two weighting passes to a keyphrase list:
|
|
10
|
+
*
|
|
11
|
+
* 1. **Wiki-entity bonus** \u2014 if any token in the phrase is a Wikipedia-linked
|
|
12
|
+
* entity (POS tag 5), the keyphrase weight is doubled and `wiki` is set.
|
|
13
|
+
* 2. **IDF domain-specificity** \u2014 weight is multiplied by the average POS
|
|
14
|
+
* uniqueness score across the phrase's tokens. Rare, domain-specific terms
|
|
15
|
+
* (high uniqueness) receive a larger multiplier than common nouns.
|
|
16
|
+
* 3. **Heavy-query bias** (optional) \u2014 if `heavyWeightQuery` is set, keyphrases
|
|
17
|
+
* that closely match the query words receive a large additive bonus, allowing
|
|
18
|
+
* dynamic re-ranking when a user clicks a term or arrives via a search query.
|
|
19
|
+
*
|
|
20
|
+
* @param keyphrases - Keyphrases to weight; mutated in-place for efficiency.
|
|
21
|
+
* @param phrasesModel - Trie model passed to the tokenizer (may be undefined).
|
|
22
|
+
* @param heavyWeightQuery - Space-separated query string to bias ranking toward.
|
|
23
|
+
* @returns The same array with updated `weight` and `wiki` fields.
|
|
24
|
+
*
|
|
25
|
+
* @example
|
|
26
|
+
* const weighted = weightKeyphrasesBySpecificity(keyphrases, model, "self attention");
|
|
27
|
+
*/
|
|
28
|
+
export function weightKeyphrasesBySpecificity(
|
|
29
|
+
keyphrases: KeyphraseEntry[],
|
|
30
|
+
phrasesModel: PhrasesModel | undefined,
|
|
31
|
+
heavyWeightQuery: string,
|
|
32
|
+
): KeyphraseEntry[] {
|
|
33
|
+
const querySplit = heavyWeightQuery ? heavyWeightQuery.split(" ") : [];
|
|
34
|
+
|
|
35
|
+
for (const kp of keyphrases) {
|
|
36
|
+
const tokens = convertTextToTokens(kp.keyphrase, { phrasesModel });
|
|
37
|
+
|
|
38
|
+
// Wiki-entity bonus: any token with category 5 doubles the weight
|
|
39
|
+
if (tokens.some(t => t[1] === 5)) {
|
|
40
|
+
kp.wiki = true;
|
|
41
|
+
kp.weight *= 2;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
// IDF domain-specificity: scale by average token uniqueness score (index [1])
|
|
45
|
+
// Tokens without a score default to 4 (mid-range common noun)
|
|
46
|
+
const idfAvg = tokens.reduce((sum, t) => sum + (t[1] ?? 4), 0) / tokens.length;
|
|
47
|
+
kp.weight = Math.floor(kp.weight * idfAvg);
|
|
48
|
+
|
|
49
|
+
// Heavy-query bias: boost phrases that closely match the query
|
|
50
|
+
if (querySplit.length) {
|
|
51
|
+
const diffWords = querySplit.filter(w => !kp.keyphrase.includes(w)).length;
|
|
52
|
+
if (diffWords < 4 && diffWords < querySplit.length - 1) {
|
|
53
|
+
kp.weight += 4000 - diffWords * 1000;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
return keyphrases;
|
|
59
|
+
}
|