extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Transformer for converting raw text queries into topic-weighted phrase tokens.
|
|
3
|
+
* Handles phrase detection, typo correction, and root word stemming.
|
|
4
|
+
*/
|
|
5
|
+
import { stemWordToRoot } from "./word-to-root-stem";
|
|
6
|
+
import { isWordCommonIgnored } from "./word-is-ignored";
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Topic token tuple shape used by downstream ranking logic:
|
|
10
|
+
* [term, termCategory, uniqueness, metadata]
|
|
11
|
+
*/
|
|
12
|
+
export type TopicToken = [string, number, number, string];
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Trie structure for phrase completion lookup keyed by first two letters, then full token.
|
|
16
|
+
*/
|
|
17
|
+
type PhraseEntry = [string | null, number, number];
|
|
18
|
+
|
|
19
|
+
export type PhrasesModel = Record<string, Record<string, PhraseEntry[]>>;
|
|
20
|
+
|
|
21
|
+
export interface ConvertTextToTokensOptions {
|
|
22
|
+
phrasesModel: PhrasesModel;
|
|
23
|
+
typosModel?: Record<string, string>;
|
|
24
|
+
checkTypos?: 0 | 1;
|
|
25
|
+
ignoreStopWords?: 0 | 1;
|
|
26
|
+
checkRootWords?: 0 | 1;
|
|
27
|
+
}
|
|
28
|
+
/**
|
|
29
|
+
* @typedef {Object} Token
|
|
30
|
+
* @property {number} termCategory - The category of the term
|
|
31
|
+
* @property {number} uniqueness - The uniqueness score of the term
|
|
32
|
+
* @property {string} term - The actual term or phrase
|
|
33
|
+
*/
|
|
34
|
+
/**
|
|
35
|
+
* ### Convert Text Query to Topic Phrase Tokens
|
|
36
|
+
* <img width="350px" src="https://i.imgur.com/NDrmSRQ.png" />
|
|
37
|
+
*
|
|
38
|
+
* Returns a list of phrases that are found in Wiki Titles/ dictionary phrases World Model
|
|
39
|
+
* that match the input phrase, or just the single word if found. Search results will be
|
|
40
|
+
* more accurate if we infer likely phrases and search for those words occuring together and
|
|
41
|
+
* not just split into words and find frequency. Examples are "white house" or "state of the art"
|
|
42
|
+
* which should be searched as a phrase but would return different context if split into words.
|
|
43
|
+
* As Led Zeppelin famously put it: \u266b "'Cause you know sometimes words have two meanings."
|
|
44
|
+
*
|
|
45
|
+
* @param {string} phrase
|
|
46
|
+
* @param {Object} [options]
|
|
47
|
+
* @param {Object} options.phrasesModel - remote model
|
|
48
|
+
* @param {Object} options.typosModel - remote model
|
|
49
|
+
* @param {number} options.checkTypos - check for typos
|
|
50
|
+
* @param {number} options.ignoreStopWords - ignore 300+ overused words
|
|
51
|
+
* @param {number} options.checkRootWords - check for word's root stem
|
|
52
|
+
* @returns {Array<{termCategory: number, uniqueness: number, term: string}>}
|
|
53
|
+
* @example
|
|
54
|
+
* const result = convertTextToTokens("The president of the united states is in the white house", { phrasesModel, typosModel });
|
|
55
|
+
* console.log(result);
|
|
56
|
+
*
|
|
57
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
58
|
+
* @category Topics
|
|
59
|
+
*/
|
|
60
|
+
export function convertTextToTokens(
|
|
61
|
+
phrase: string,
|
|
62
|
+
options: Partial<ConvertTextToTokensOptions> = {},
|
|
63
|
+
): TopicToken[] {
|
|
64
|
+
let {
|
|
65
|
+
phrasesModel, //pass in remote model
|
|
66
|
+
typosModel,
|
|
67
|
+
checkTypos = 0,
|
|
68
|
+
checkRootWords = 1,
|
|
69
|
+
ignoreStopWords = 1,
|
|
70
|
+
} = options;
|
|
71
|
+
|
|
72
|
+
if (!phrasesModel) throw new Error("Missing phrasesModel ");
|
|
73
|
+
|
|
74
|
+
if (!phrase) throw new Error("Missing phrase");
|
|
75
|
+
|
|
76
|
+
//strip non-alphanumeric characters from query an keep -'/
|
|
77
|
+
phrase = phrase.replace(/[^a-zA-Z0-9\s\-\'\/]/g, " ");
|
|
78
|
+
|
|
79
|
+
//split into words
|
|
80
|
+
var words = phrase.toLowerCase().split(/\W+/);
|
|
81
|
+
|
|
82
|
+
//check for typos
|
|
83
|
+
|
|
84
|
+
if (checkTypos && typosModel)
|
|
85
|
+
words = words.map((word) => typosModel[word] || word);
|
|
86
|
+
|
|
87
|
+
const topics: TopicToken[] = [];
|
|
88
|
+
for (var i = 0; i < words.length; i++) {
|
|
89
|
+
var word = words[i];
|
|
90
|
+
|
|
91
|
+
//ignore 300+ common stop words
|
|
92
|
+
if (ignoreStopWords && isWordCommonIgnored(word)) {
|
|
93
|
+
//todo include as ignored
|
|
94
|
+
topics.push([word, 0, 0, ""]);
|
|
95
|
+
continue;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
//Find next word phrase completion list
|
|
99
|
+
var firstTwoLetters = word.slice(0, 2);
|
|
100
|
+
var possiblePhrases = phrasesModel[firstTwoLetters]
|
|
101
|
+
? phrasesModel[firstTwoLetters][word]
|
|
102
|
+
: null;
|
|
103
|
+
|
|
104
|
+
//check for root words like "gaming" -> "game"
|
|
105
|
+
if (!possiblePhrases && checkRootWords) {
|
|
106
|
+
var rootWord = stemWordToRoot(word);
|
|
107
|
+
if (rootWord !== word)
|
|
108
|
+
possiblePhrases = phrasesModel[rootWord.slice(0, 2)]
|
|
109
|
+
? phrasesModel[rootWord.slice(0, 2)][rootWord]
|
|
110
|
+
: null;
|
|
111
|
+
|
|
112
|
+
//TODO add label "root word"
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
//if word still not in dict, add it as a single word
|
|
116
|
+
if (!possiblePhrases) topics.push([word, 0, 0, ""]);
|
|
117
|
+
|
|
118
|
+
if (possiblePhrases) {
|
|
119
|
+
var maxPhraseLength = 1;
|
|
120
|
+
let singleWordObj: TopicToken | null = null;
|
|
121
|
+
var isPhraseFound = false;
|
|
122
|
+
|
|
123
|
+
//calculate max possible length of phrase of next words
|
|
124
|
+
for (var p of possiblePhrases)
|
|
125
|
+
if (p[0]?.length > maxPhraseLength) maxPhraseLength = p[0].length;
|
|
126
|
+
|
|
127
|
+
//grab that length of text from next words
|
|
128
|
+
var nextWords = "";
|
|
129
|
+
for (var j = 1; j < words.length - i; j++) {
|
|
130
|
+
nextWords += (words[i + j] || "") + " ";
|
|
131
|
+
|
|
132
|
+
if (nextWords.length >= maxPhraseLength) break;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
for (const phraseEntry of possiblePhrases) {
|
|
136
|
+
//if no next phrase, preserve the single word
|
|
137
|
+
//it culd also be not in the dict first word
|
|
138
|
+
if (!phraseEntry[0] && !singleWordObj) {
|
|
139
|
+
const nextPhrase = phraseEntry.slice(1, 3) as [number, number];
|
|
140
|
+
singleWordObj = [word, nextPhrase[0], nextPhrase[1], ""];
|
|
141
|
+
} else {
|
|
142
|
+
//add next word to the phrase up to maxPhraseLength
|
|
143
|
+
const nextPhrasePart = phraseEntry[0];
|
|
144
|
+
if (
|
|
145
|
+
!isPhraseFound &&
|
|
146
|
+
typeof nextPhrasePart === "string" &&
|
|
147
|
+
nextWords.startsWith(nextPhrasePart)
|
|
148
|
+
) {
|
|
149
|
+
var fullPhrase = word + " " + nextPhrasePart;
|
|
150
|
+
const nextPhrase = phraseEntry.slice(1, 3) as [number, number];
|
|
151
|
+
topics.push([fullPhrase, nextPhrase[0], nextPhrase[1], ""]); // remove first which is next words
|
|
152
|
+
|
|
153
|
+
//skip looping thru the next words added to phrase
|
|
154
|
+
i += nextPhrasePart.split(" ").length; //TODO fi
|
|
155
|
+
|
|
156
|
+
//suppress single-word "red" if "red wine" is found
|
|
157
|
+
isPhraseFound = true;
|
|
158
|
+
|
|
159
|
+
break;
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
//if no phrases then add the single word
|
|
165
|
+
if (!isPhraseFound) {
|
|
166
|
+
if (singleWordObj) {
|
|
167
|
+
// Could be starter token for phrases; keep it as standalone suggestion.
|
|
168
|
+
topics.push(singleWordObj);
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
return topics.filter(Boolean);
|
|
175
|
+
}
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Utility for identifying common stop words (e.g., "the", "and", "is").
|
|
3
|
+
* Based on the SpaCy English stop word list.
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Checks word is in [320 commonly ignored "stop words
|
|
8
|
+
* "](https://raw.githubusercontent.com/igorbrigadir/stopwords/master/en/spacy.txt)
|
|
9
|
+
* in queries, using efficient JS Set method
|
|
10
|
+
* @param {string} word
|
|
11
|
+
* @returns {Boolean}
|
|
12
|
+
*/
|
|
13
|
+
export function isWordCommonIgnored(word) {
|
|
14
|
+
return stopwords.has(word.toLowerCase());
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
//@ts-ignore
|
|
18
|
+
const stopwords = new Set([
|
|
19
|
+
"a", "about", "above", "across", "after", "afterwards", "again", "against", "all",
|
|
20
|
+
"almost", "alone", "along", "already", "also", "although", "always", "am", "among",
|
|
21
|
+
"amongst", "amount", "an", "and", "another", "any", "anyhow", "anyone", "anything",
|
|
22
|
+
"anyway", "anywhere", "are", "around", "as", "at", "back", "be", "became", "because",
|
|
23
|
+
"become", "becomes", "becoming", "been", "before", "beforehand", "behind", "being",
|
|
24
|
+
"below", "beside", "besides", "between", "beyond", "both", "bottom", "but", "by",
|
|
25
|
+
"ca", "call", "can", "cannot", "could", "did", "do", "does", "doing", "done", "down",
|
|
26
|
+
"due", "during", "each", "eight", "either", "eleven", "else", "elsewhere", "empty",
|
|
27
|
+
"enough", "even", "ever", "every", "everyone", "everything", "everywhere", "except",
|
|
28
|
+
"few", "fifteen", "fifty", "first", "five", "for", "former", "formerly", "forty",
|
|
29
|
+
"four", "from", "front", "full", "further", "get", "give", "go", "had", "has",
|
|
30
|
+
"have", "he", "hence", "her", "here", "hereafter", "hereby", "herein", "hereupon",
|
|
31
|
+
"hers", "herself", "him", "himself", "his", "how", "however", "hundred", "i", "if",
|
|
32
|
+
"in", "indeed", "into", "is", "it", "its", "itself", "just", "keep", "last", "latter",
|
|
33
|
+
"latterly", "least", "less", "made", "make", "many", "may", "me", "meanwhile",
|
|
34
|
+
"might", "mine", "more", "moreover", "most", "mostly", "move", "much", "must", "my",
|
|
35
|
+
"myself", "n't", "name", "namely", "neither", "never", "nevertheless", "next", "nine",
|
|
36
|
+
"no", "nobody", "none", "noone", "nor", "not", "nothing", "now", "nowhere", "of",
|
|
37
|
+
"off", "often", "on", "once", "one", "only", "onto", "or", "other", "others",
|
|
38
|
+
"otherwise", "our", "ours", "ourselves", "out", "over", "own", "part", "per",
|
|
39
|
+
"perhaps", "please", "put", "quite", "rather", "re", "really", "regarding", "same",
|
|
40
|
+
"say", "see", "seem", "seemed", "seeming", "seems", "serious", "several", "she",
|
|
41
|
+
"should", "show", "side", "since", "six", "sixty", "so", "some", "somehow",
|
|
42
|
+
"someone", "something", "sometime", "sometimes", "somewhere", "still", "such",
|
|
43
|
+
"take", "ten", "than", "that", "the", "their", "them", "themselves", "then",
|
|
44
|
+
"thence", "there", "thereafter", "thereby", "therefore", "therein", "thereupon",
|
|
45
|
+
"these", "they", "third", "this", "those", "though", "three", "through", "throughout",
|
|
46
|
+
"thru", "thus", "to", "together", "too", "top", "toward", "towards", "twelve", "twenty",
|
|
47
|
+
"two", "under", "unless", "until", "up", "upon", "us", "used", "using", "various",
|
|
48
|
+
"very", "via", "was", "we", "well", "were", "what", "whatever", "when", "whence",
|
|
49
|
+
"whenever", "where", "whereafter", "whereas", "whereby", "wherein", "whereupon",
|
|
50
|
+
"wherever", "whether", "which", "while", "whither", "who", "whoever", "whole",
|
|
51
|
+
"whom", "whose", "why", "will", "with", "within", "without", "would", "yet", "you",
|
|
52
|
+
"your", "yours", "yourself", "yourselves", "'d", "'ll", "'m", "'re", "'s", "'ve"
|
|
53
|
+
])
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Implementation of the Porter Stemmer algorithm for word normalization.
|
|
3
|
+
* Used to reduce words to their root form (e.g., "running" to "run").
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Stems a word using the [Porter
|
|
7
|
+
* Stemmer](https://snowballstem.org/algorithms/porter/stemmer.html)
|
|
8
|
+
* for removing inflectional endings like "ing", "ist", "ize".
|
|
9
|
+
*
|
|
10
|
+
* @author [Porter, M. (1980)](https://tartarus.org/martin/PorterStemmer/)
|
|
11
|
+
* @param word - The word to be stemmed
|
|
12
|
+
* @returns The stemmed word
|
|
13
|
+
* @example const rootWord = stemWordToRoot("running"); // returns "run"
|
|
14
|
+
* @category Topics
|
|
15
|
+
*/
|
|
16
|
+
export function stemWordToRoot(word: string): string {
|
|
17
|
+
// Return short words (less than 3 characters) without stemming
|
|
18
|
+
if (word.length < 3) return word;
|
|
19
|
+
|
|
20
|
+
const SUFFIX_MAPS = {
|
|
21
|
+
step2: {
|
|
22
|
+
ational: "ate",
|
|
23
|
+
tional: "tion",
|
|
24
|
+
enci: "ence",
|
|
25
|
+
anci: "ance",
|
|
26
|
+
izer: "ize",
|
|
27
|
+
bli: "ble",
|
|
28
|
+
alli: "al",
|
|
29
|
+
entli: "ent",
|
|
30
|
+
eli: "e",
|
|
31
|
+
ousli: "ous",
|
|
32
|
+
ization: "ize",
|
|
33
|
+
ation: "ate",
|
|
34
|
+
ator: "ate",
|
|
35
|
+
alism: "al",
|
|
36
|
+
iveness: "ive",
|
|
37
|
+
fulness: "ful",
|
|
38
|
+
ousness: "ous",
|
|
39
|
+
aliti: "al",
|
|
40
|
+
iviti: "ive",
|
|
41
|
+
biliti: "ble",
|
|
42
|
+
logi: "log",
|
|
43
|
+
},
|
|
44
|
+
step3: {
|
|
45
|
+
icate: "ic",
|
|
46
|
+
ative: "",
|
|
47
|
+
alize: "al",
|
|
48
|
+
iciti: "ic",
|
|
49
|
+
ical: "ic",
|
|
50
|
+
ful: "",
|
|
51
|
+
ness: "",
|
|
52
|
+
},
|
|
53
|
+
};
|
|
54
|
+
|
|
55
|
+
const REGEX = {
|
|
56
|
+
consonant: /[^aeiou]/,
|
|
57
|
+
vowel: /[aeiouy]/,
|
|
58
|
+
consonants: /([^aeiou][^aeiouy]*)/,
|
|
59
|
+
vowels: /([aeiouy][aeiou]*)/,
|
|
60
|
+
gt0: /^([^aeiou][^aeiouy]*)?([aeiouy][aeiou]*)([^aeiou][^aeiouy]*)/,
|
|
61
|
+
eq1: /^([^aeiou][^aeiouy]*)?([aeiouy][aeiou]*)([^aeiou][^aeiouy]*)([aeiouy][aeiou]*)?$/,
|
|
62
|
+
gt1: /^([^aeiou][^aeiouy]*)?([aeiouy][aeiou]*[^aeiou][^aeiouy]*){2,}/,
|
|
63
|
+
vowelInStem: /^([^aeiou][^aeiouy]*)?[aeiouy]/,
|
|
64
|
+
consonantLike: /^([^aeiou][^aeiouy]*)[aeiouy][^aeiouwxy]$/,
|
|
65
|
+
sfxLl: /ll$/,
|
|
66
|
+
sfxE: /^(.+?)e$/,
|
|
67
|
+
sfxY: /^(.+?)y$/,
|
|
68
|
+
sfxIon: /^(.+?(s|t))(ion)$/,
|
|
69
|
+
sfxEdOrIng: /^(.+?)(ed|ing)$/,
|
|
70
|
+
sfxAtOrBlOrIz: /(at|bl|iz)$/,
|
|
71
|
+
sfxEED: /^(.+?)eed$/,
|
|
72
|
+
sfxS: /^.+?[^s]s$/,
|
|
73
|
+
sfxSsesOrIes: /^.+?(ss|i)es$/,
|
|
74
|
+
sfxMultiConsonantLike: /([^aeiouylsz])\1$/,
|
|
75
|
+
step2:
|
|
76
|
+
/^(.+?)(ational|tional|enci|anci|izer|bli|alli|entli|eli|ousli|ization|ation|ator|alism|iveness|fulness|ousness|aliti|iviti|biliti|logi)$/,
|
|
77
|
+
step3: /^(.+?)(icate|ative|alize|iciti|ical|ful|ness)$/,
|
|
78
|
+
step4:
|
|
79
|
+
/^(.+?)(al|ance|ence|er|ic|able|ible|ant|ement|ment|ent|ou|ism|ate|iti|ous|ive|ize)$/,
|
|
80
|
+
};
|
|
81
|
+
|
|
82
|
+
let stem = word.toLowerCase();
|
|
83
|
+
|
|
84
|
+
// Special handling for words starting with 'y'
|
|
85
|
+
const firstCharWasY = stem[0] === "y";
|
|
86
|
+
if (firstCharWasY) stem = "Y" + stem.slice(1);
|
|
87
|
+
|
|
88
|
+
// Step 1a: Handle plural forms and -ed or -ing suffixes
|
|
89
|
+
stem = stem
|
|
90
|
+
.replace(REGEX.sfxSsesOrIes, "$1")
|
|
91
|
+
.replace(REGEX.sfxS, (s) => s.slice(0, -1));
|
|
92
|
+
|
|
93
|
+
// Step 1b: Handle -eed, -ed, -ing suffixes
|
|
94
|
+
let match;
|
|
95
|
+
if ((match = REGEX.sfxEED.exec(stem))) {
|
|
96
|
+
if (REGEX.gt0.test(match[1])) stem = stem.slice(0, -1);
|
|
97
|
+
} else if (
|
|
98
|
+
(match = REGEX.sfxEdOrIng.exec(stem)) &&
|
|
99
|
+
REGEX.vowelInStem.test(match[1])
|
|
100
|
+
) {
|
|
101
|
+
stem = match[1];
|
|
102
|
+
if (REGEX.sfxAtOrBlOrIz.test(stem)) {
|
|
103
|
+
stem += "e";
|
|
104
|
+
} else if (REGEX.sfxMultiConsonantLike.test(stem)) {
|
|
105
|
+
stem = stem.slice(0, -1);
|
|
106
|
+
} else if (REGEX.consonantLike.test(stem)) {
|
|
107
|
+
stem += "e";
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
// Step 1c: Replace suffix y or Y by i if preceded by a non-vowel
|
|
112
|
+
if ((match = REGEX.sfxY.exec(stem)) && REGEX.vowelInStem.test(match[1])) {
|
|
113
|
+
stem = match[1] + "i";
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
// Step 2: Handle various suffixes
|
|
117
|
+
if ((match = REGEX.step2.exec(stem)) && REGEX.gt0.test(match[1])) {
|
|
118
|
+
stem = match[1] + SUFFIX_MAPS.step2[match[2]];
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
// Step 3: Handle more suffixes
|
|
122
|
+
if ((match = REGEX.step3.exec(stem)) && REGEX.gt0.test(match[1])) {
|
|
123
|
+
stem = match[1] + SUFFIX_MAPS.step3[match[2]];
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
// Step 4: Handle even more suffixes
|
|
127
|
+
if ((match = REGEX.step4.exec(stem))) {
|
|
128
|
+
if (REGEX.gt1.test(match[1])) stem = match[1];
|
|
129
|
+
} else if ((match = REGEX.sfxIon.exec(stem)) && REGEX.gt1.test(match[1])) {
|
|
130
|
+
stem = match[1];
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
// Step 5a: Remove final e
|
|
134
|
+
if (
|
|
135
|
+
(match = REGEX.sfxE.exec(stem)) &&
|
|
136
|
+
(REGEX.gt1.test(match[1]) ||
|
|
137
|
+
(REGEX.eq1.test(match[1]) && !REGEX.consonantLike.test(match[1])))
|
|
138
|
+
) {
|
|
139
|
+
stem = match[1];
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
// Step 5b: Remove final ll
|
|
143
|
+
if (REGEX.sfxLl.test(stem) && REGEX.gt1.test(stem)) {
|
|
144
|
+
stem = stem.slice(0, -1);
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
// Restore initial Y if it was changed
|
|
148
|
+
if (firstCharWasY) stem = "y" + stem.slice(1);
|
|
149
|
+
|
|
150
|
+
return stem;
|
|
151
|
+
}
|
package/src/types.d.ts
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
import { Model } from "./models/types";
|
|
2
|
+
|
|
3
|
+
type BaseUIConfigField = {
|
|
4
|
+
name: string;
|
|
5
|
+
key: string;
|
|
6
|
+
required: boolean;
|
|
7
|
+
description: string;
|
|
8
|
+
scope: "client" | "server";
|
|
9
|
+
env?: string;
|
|
10
|
+
};
|
|
11
|
+
|
|
12
|
+
type StringUIConfigField = BaseUIConfigField & {
|
|
13
|
+
type: "string";
|
|
14
|
+
placeholder?: string;
|
|
15
|
+
default?: string;
|
|
16
|
+
};
|
|
17
|
+
|
|
18
|
+
type SelectUIConfigFieldOptions = {
|
|
19
|
+
name: string;
|
|
20
|
+
value: string;
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
type SelectUIConfigField = BaseUIConfigField & {
|
|
24
|
+
type: "select";
|
|
25
|
+
default?: string;
|
|
26
|
+
options: SelectUIConfigFieldOptions[];
|
|
27
|
+
};
|
|
28
|
+
|
|
29
|
+
type PasswordUIConfigField = BaseUIConfigField & {
|
|
30
|
+
type: "password";
|
|
31
|
+
placeholder?: string;
|
|
32
|
+
default?: string;
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
type TextareaUIConfigField = BaseUIConfigField & {
|
|
36
|
+
type: "textarea";
|
|
37
|
+
placeholder?: string;
|
|
38
|
+
default?: string;
|
|
39
|
+
};
|
|
40
|
+
|
|
41
|
+
type SwitchUIConfigField = BaseUIConfigField & {
|
|
42
|
+
type: "switch";
|
|
43
|
+
default?: boolean;
|
|
44
|
+
};
|
|
45
|
+
|
|
46
|
+
type UIConfigField =
|
|
47
|
+
| StringUIConfigField
|
|
48
|
+
| SelectUIConfigField
|
|
49
|
+
| PasswordUIConfigField
|
|
50
|
+
| TextareaUIConfigField
|
|
51
|
+
| SwitchUIConfigField;
|
|
52
|
+
|
|
53
|
+
type ConfigModelProvider = {
|
|
54
|
+
id: string;
|
|
55
|
+
name: string;
|
|
56
|
+
type: string;
|
|
57
|
+
chatModels: Model[];
|
|
58
|
+
config: { [key: string]: any };
|
|
59
|
+
hash: string;
|
|
60
|
+
isEnvBased?: boolean; // True if provider was created from environment variables (global keys)
|
|
61
|
+
};
|
|
62
|
+
|
|
63
|
+
type MCPServerConfig = {
|
|
64
|
+
id: string;
|
|
65
|
+
name: string;
|
|
66
|
+
type: string;
|
|
67
|
+
url?: string;
|
|
68
|
+
apiKey?: string;
|
|
69
|
+
config: { [key: string]: any };
|
|
70
|
+
enabled: boolean;
|
|
71
|
+
hash: string;
|
|
72
|
+
};
|
|
73
|
+
|
|
74
|
+
type Config = {
|
|
75
|
+
version: number;
|
|
76
|
+
setupComplete: boolean;
|
|
77
|
+
preferences: {
|
|
78
|
+
[key: string]: any;
|
|
79
|
+
};
|
|
80
|
+
personalization: {
|
|
81
|
+
[key: string]: any;
|
|
82
|
+
};
|
|
83
|
+
modelProviders: ConfigModelProvider[];
|
|
84
|
+
mcpServers: MCPServerConfig[];
|
|
85
|
+
search: {
|
|
86
|
+
[key: string]: any;
|
|
87
|
+
};
|
|
88
|
+
};
|
|
89
|
+
|
|
90
|
+
type EnvMap = {
|
|
91
|
+
[key: string]: {
|
|
92
|
+
fieldKey: string;
|
|
93
|
+
providerKey: string;
|
|
94
|
+
};
|
|
95
|
+
};
|
|
96
|
+
|
|
97
|
+
type ModelProviderUISection = {
|
|
98
|
+
name: string;
|
|
99
|
+
key: string;
|
|
100
|
+
fields: UIConfigField[];
|
|
101
|
+
};
|
|
102
|
+
|
|
103
|
+
type MCPServerUISection = {
|
|
104
|
+
name: string;
|
|
105
|
+
key: string;
|
|
106
|
+
fields: UIConfigField[];
|
|
107
|
+
};
|
|
108
|
+
|
|
109
|
+
type UIConfigSections = {
|
|
110
|
+
preferences: UIConfigField[];
|
|
111
|
+
personalization: UIConfigField[];
|
|
112
|
+
modelProviders: ModelProviderUISection[];
|
|
113
|
+
mcpServers: MCPServerUISection[];
|
|
114
|
+
search: UIConfigField[];
|
|
115
|
+
};
|
|
116
|
+
|
|
117
|
+
export type {
|
|
118
|
+
UIConfigField,
|
|
119
|
+
Config,
|
|
120
|
+
EnvMap,
|
|
121
|
+
UIConfigSections,
|
|
122
|
+
SelectUIConfigField,
|
|
123
|
+
StringUIConfigField,
|
|
124
|
+
ModelProviderUISection,
|
|
125
|
+
MCPServerUISection,
|
|
126
|
+
ConfigModelProvider,
|
|
127
|
+
MCPServerConfig,
|
|
128
|
+
TextareaUIConfigField,
|
|
129
|
+
SwitchUIConfigField,
|
|
130
|
+
};
|