@sideid/id-profanity-filter 1.9.5 → 1.10.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.eslintrc.js +44 -16
- package/.github/workflows/release.yml +62 -0
- package/CONTRIBUTING.md +150 -150
- package/LICENSE +21 -21
- package/README.md +548 -285
- package/dist/config/options.d.ts +24 -0
- package/dist/constants/categories/blasphemy.d.ts +4 -0
- package/dist/constants/categories/disgusting.d.ts +4 -0
- package/dist/constants/categories/drugs.d.ts +4 -0
- package/dist/constants/categories/profanity.d.ts +4 -0
- package/dist/constants/categories/slur.d.ts +4 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.esm.js +923 -94
- package/dist/index.esm.js.map +1 -1
- package/dist/index.js +924 -93
- package/dist/index.js.map +1 -1
- package/dist/types/index.d.ts +2 -0
- package/dist/utils/ahoCorasick.d.ts +36 -0
- package/dist/utils/similarityUtils.d.ts +35 -0
- package/eslint.config.mjs +40 -0
- package/examples/advanced.ts +120 -0
- package/examples/basic.ts +71 -52
- package/examples/custom-list.ts +140 -0
- package/jest.config.mjs +10 -10
- package/package.json +3 -2
- package/prettierrc +6 -6
- package/rollup.config.mjs +35 -35
- package/src/config/options.ts +2 -0
- package/src/constants/categories/blasphemy.ts +25 -0
- package/src/constants/categories/disgusting.ts +82 -0
- package/src/constants/categories/drugs.ts +72 -0
- package/src/constants/categories/profanity.ts +139 -0
- package/src/constants/categories/slur.ts +102 -0
- package/src/constants/regions/general.ts +111 -2
- package/src/constants/regions/jawa.ts +257 -3
- package/src/constants/wordList.ts +15 -8
- package/src/core/analyzer.ts +28 -13
- package/src/core/filter.ts +178 -37
- package/src/core/matcher.ts +146 -69
- package/src/index.ts +21 -2
- package/src/types/index.ts +4 -2
- package/src/utils/ahoCorasick.ts +179 -0
- package/src/utils/regexUtils.ts +0 -1
- package/src/utils/similarityUtils.ts +239 -7
- package/tsconfig.json +115 -115
- package/.github/workflows/ci.yml +0 -0
- package/src/constants/categories/index.ts +0 -31
- package/src/constants/regions/index.ts +0 -62
package/src/core/filter.ts
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
import { FilterOptions, FilterResult, ProfanityWord } from "../types";
|
|
2
2
|
import { findProfanity, findProfanityWithMetadata } from "./matcher";
|
|
3
|
-
import { censorWord, escapeRegExp
|
|
3
|
+
import { censorWord, escapeRegExp } from "../utils/stringUtils";
|
|
4
4
|
import { createWordRegex } from "../utils/regexUtils";
|
|
5
|
-
import {
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
5
|
+
import { DEFAULT_OPTIONS, makeRandomGrawlixString } from "../config/options";
|
|
6
|
+
|
|
7
|
+
interface FindProfanityFunction {
|
|
8
|
+
(text: string, options?: FilterOptions): string[];
|
|
9
|
+
lastActualMatches?: Map<string, string[]>;
|
|
10
|
+
}
|
|
10
11
|
|
|
11
12
|
/**
|
|
12
13
|
* Menyensor kata kotor dalam teks
|
|
@@ -28,6 +29,11 @@ export function filter(
|
|
|
28
29
|
useRandomGrawlix = false,
|
|
29
30
|
keepFirstAndLast = false,
|
|
30
31
|
indonesianVariation = false,
|
|
32
|
+
detectSplit = false,
|
|
33
|
+
detectSimilarity = false,
|
|
34
|
+
useLevenshtein = false,
|
|
35
|
+
maxLevenshteinDistance = 2,
|
|
36
|
+
similarityThreshold = 0.8,
|
|
31
37
|
} = { ...DEFAULT_OPTIONS, ...options };
|
|
32
38
|
|
|
33
39
|
const matches = findProfanity(text, {
|
|
@@ -36,8 +42,16 @@ export function filter(
|
|
|
36
42
|
whitelist,
|
|
37
43
|
checkSubstring,
|
|
38
44
|
indonesianVariation,
|
|
45
|
+
detectSplit,
|
|
46
|
+
detectSimilarity,
|
|
47
|
+
useLevenshtein,
|
|
48
|
+
maxLevenshteinDistance,
|
|
49
|
+
similarityThreshold,
|
|
39
50
|
});
|
|
40
51
|
|
|
52
|
+
const actualMatches: Map<string, string[]> =
|
|
53
|
+
(findProfanity as FindProfanityFunction).lastActualMatches || new Map();
|
|
54
|
+
|
|
41
55
|
const matchDetails = findProfanityWithMetadata(text, options);
|
|
42
56
|
|
|
43
57
|
if (matches.length === 0) {
|
|
@@ -66,49 +80,176 @@ export function filter(
|
|
|
66
80
|
)),
|
|
67
81
|
);
|
|
68
82
|
|
|
69
|
-
const
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
detectSplit: false,
|
|
74
|
-
indonesianVariation: false,
|
|
75
|
-
});
|
|
83
|
+
const variants = actualMatches.get(word.toLowerCase()) || [];
|
|
84
|
+
variants.push(word);
|
|
85
|
+
|
|
86
|
+
const uniqueVariants = [...new Set(variants)];
|
|
76
87
|
|
|
77
|
-
|
|
78
|
-
|
|
88
|
+
uniqueVariants.forEach((variant) => {
|
|
89
|
+
const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
|
|
79
90
|
|
|
80
|
-
|
|
91
|
+
let match;
|
|
92
|
+
while ((match = regex.exec(filteredText)) !== null) {
|
|
93
|
+
const originalWord = match[0];
|
|
81
94
|
|
|
82
|
-
|
|
83
|
-
const originalWord = match[0];
|
|
95
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
84
96
|
|
|
85
|
-
|
|
97
|
+
let censoredWord;
|
|
98
|
+
if (useRandomGrawlix) {
|
|
99
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
100
|
+
} else {
|
|
101
|
+
censoredWord = censorWord(
|
|
102
|
+
originalWord,
|
|
103
|
+
replaceWith,
|
|
104
|
+
!fullWordCensor && keepFirstAndLast,
|
|
105
|
+
);
|
|
106
|
+
}
|
|
86
107
|
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
108
|
+
replacements.push({
|
|
109
|
+
original: originalWord,
|
|
110
|
+
censored: censoredWord,
|
|
111
|
+
metadata,
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
filteredText = filteredText.replace(
|
|
115
|
+
new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"),
|
|
116
|
+
censoredWord,
|
|
95
117
|
);
|
|
96
118
|
}
|
|
119
|
+
});
|
|
97
120
|
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
121
|
+
if (detectSplit || detectLeetSpeak) {
|
|
122
|
+
if (detectLeetSpeak) {
|
|
123
|
+
const leetRegex = createWordRegex(word, {
|
|
124
|
+
wholeWord: true,
|
|
125
|
+
caseSensitive: false,
|
|
126
|
+
leetSpeak: true,
|
|
127
|
+
detectSplit: false,
|
|
128
|
+
indonesianVariation: false,
|
|
129
|
+
});
|
|
103
130
|
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
131
|
+
let match;
|
|
132
|
+
while ((match = leetRegex.exec(filteredText)) !== null) {
|
|
133
|
+
const originalWord = match[0];
|
|
134
|
+
|
|
135
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
136
|
+
|
|
137
|
+
let censoredWord;
|
|
138
|
+
if (useRandomGrawlix) {
|
|
139
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
140
|
+
} else {
|
|
141
|
+
censoredWord = censorWord(
|
|
142
|
+
originalWord,
|
|
143
|
+
replaceWith,
|
|
144
|
+
!fullWordCensor && keepFirstAndLast,
|
|
145
|
+
);
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
replacements.push({
|
|
149
|
+
original: originalWord,
|
|
150
|
+
censored: censoredWord,
|
|
151
|
+
metadata,
|
|
152
|
+
});
|
|
153
|
+
|
|
154
|
+
filteredText = filteredText.replace(
|
|
155
|
+
new RegExp(escapeRegExp(originalWord), "g"),
|
|
156
|
+
censoredWord,
|
|
157
|
+
);
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
if (detectSplit) {
|
|
162
|
+
const splitRegex = createWordRegex(word, {
|
|
163
|
+
wholeWord: false,
|
|
164
|
+
caseSensitive: false,
|
|
165
|
+
leetSpeak: false,
|
|
166
|
+
detectSplit: true,
|
|
167
|
+
indonesianVariation: false,
|
|
168
|
+
});
|
|
169
|
+
|
|
170
|
+
let match;
|
|
171
|
+
while ((match = splitRegex.exec(filteredText)) !== null) {
|
|
172
|
+
const originalWord = match[0];
|
|
173
|
+
|
|
174
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
175
|
+
|
|
176
|
+
let censoredWord;
|
|
177
|
+
if (useRandomGrawlix) {
|
|
178
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
179
|
+
} else {
|
|
180
|
+
censoredWord = censorWord(
|
|
181
|
+
originalWord,
|
|
182
|
+
replaceWith,
|
|
183
|
+
!fullWordCensor && keepFirstAndLast,
|
|
184
|
+
);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
replacements.push({
|
|
188
|
+
original: originalWord,
|
|
189
|
+
censored: censoredWord,
|
|
190
|
+
metadata,
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
filteredText = filteredText.replace(
|
|
194
|
+
new RegExp(escapeRegExp(originalWord), "g"),
|
|
195
|
+
censoredWord,
|
|
196
|
+
);
|
|
197
|
+
}
|
|
198
|
+
}
|
|
109
199
|
}
|
|
110
200
|
});
|
|
111
201
|
|
|
202
|
+
if (detectSimilarity && useLevenshtein) {
|
|
203
|
+
matches.forEach((word) => {
|
|
204
|
+
const metadata = matchDetails.find(
|
|
205
|
+
(m) =>
|
|
206
|
+
m.word.toLowerCase() === word.toLowerCase() ||
|
|
207
|
+
(m.aliases &&
|
|
208
|
+
m.aliases.some(
|
|
209
|
+
(alias) => alias.toLowerCase() === word.toLowerCase(),
|
|
210
|
+
)),
|
|
211
|
+
);
|
|
212
|
+
|
|
213
|
+
const variants = actualMatches.get(word.toLowerCase()) || [];
|
|
214
|
+
|
|
215
|
+
variants.forEach((variant) => {
|
|
216
|
+
const exactVariantRegex = new RegExp(
|
|
217
|
+
`\\b${escapeRegExp(variant)}\\b`,
|
|
218
|
+
"gi",
|
|
219
|
+
);
|
|
220
|
+
|
|
221
|
+
let match;
|
|
222
|
+
while ((match = exactVariantRegex.exec(filteredText)) !== null) {
|
|
223
|
+
const originalWord = match[0];
|
|
224
|
+
|
|
225
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
226
|
+
|
|
227
|
+
let censoredWord;
|
|
228
|
+
if (useRandomGrawlix) {
|
|
229
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
230
|
+
} else {
|
|
231
|
+
censoredWord = censorWord(
|
|
232
|
+
originalWord,
|
|
233
|
+
replaceWith,
|
|
234
|
+
!fullWordCensor && keepFirstAndLast,
|
|
235
|
+
);
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
replacements.push({
|
|
239
|
+
original: originalWord,
|
|
240
|
+
censored: censoredWord,
|
|
241
|
+
metadata,
|
|
242
|
+
});
|
|
243
|
+
|
|
244
|
+
filteredText = filteredText.replace(
|
|
245
|
+
new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"),
|
|
246
|
+
censoredWord,
|
|
247
|
+
);
|
|
248
|
+
}
|
|
249
|
+
});
|
|
250
|
+
});
|
|
251
|
+
}
|
|
252
|
+
|
|
112
253
|
return {
|
|
113
254
|
filtered: filteredText,
|
|
114
255
|
censored: replacements.length,
|
package/src/core/matcher.ts
CHANGED
|
@@ -5,24 +5,34 @@ import {
|
|
|
5
5
|
FilterOptions,
|
|
6
6
|
} from "../types";
|
|
7
7
|
|
|
8
|
-
import { wordObjects
|
|
9
|
-
import {
|
|
10
|
-
|
|
11
|
-
escapeRegExp,
|
|
12
|
-
containsAnyWord,
|
|
13
|
-
detectSplitWords,
|
|
14
|
-
} from "../utils/stringUtils";
|
|
15
|
-
import {
|
|
16
|
-
createWordRegex,
|
|
17
|
-
addLeetSpeakVariations,
|
|
18
|
-
addIndonesianVariations,
|
|
19
|
-
addSplitVariations,
|
|
20
|
-
} from "../utils/regexUtils";
|
|
8
|
+
import { wordObjects } from "../constants/wordList";
|
|
9
|
+
import { normalizeText } from "../utils/stringUtils";
|
|
10
|
+
import { createWordRegex } from "../utils/regexUtils";
|
|
21
11
|
import {
|
|
22
12
|
findPossibleProfanityBySimiliarity,
|
|
23
|
-
|
|
13
|
+
findProfanityByLevenshteinDistance,
|
|
24
14
|
} from "../utils/similarityUtils";
|
|
25
15
|
import { DEFAULT_OPTIONS } from "../config/options";
|
|
16
|
+
import { AhoCorasick } from "../utils/ahoCorasick";
|
|
17
|
+
|
|
18
|
+
const globalAhoCorasick = new AhoCorasick();
|
|
19
|
+
let ahoCorasickInitialized = false;
|
|
20
|
+
|
|
21
|
+
function initializeAhoCorasick(words: string[]) {
|
|
22
|
+
if (ahoCorasickInitialized) return;
|
|
23
|
+
|
|
24
|
+
for (const word of words) {
|
|
25
|
+
globalAhoCorasick.addPattern(word);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
globalAhoCorasick.build();
|
|
29
|
+
ahoCorasickInitialized = true;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
interface FindProfanityFunction {
|
|
33
|
+
(text: string, options?: FilterOptions): string[];
|
|
34
|
+
lastActualMatches?: Map<string, string[]>;
|
|
35
|
+
}
|
|
26
36
|
|
|
27
37
|
export function findProfanity(
|
|
28
38
|
text: string,
|
|
@@ -40,53 +50,69 @@ export function findProfanity(
|
|
|
40
50
|
detectSimilarity = false,
|
|
41
51
|
similarityThreshold = 0.8,
|
|
42
52
|
detectSplit = false,
|
|
53
|
+
useLevenshtein = false,
|
|
54
|
+
maxLevenshteinDistance = 2,
|
|
43
55
|
} = { ...DEFAULT_OPTIONS, ...options };
|
|
44
56
|
|
|
45
57
|
const normalizedText = normalizeText(text);
|
|
46
58
|
|
|
47
|
-
let
|
|
59
|
+
let baseWordsToCheck: string[] = wordList.length > 0 ? wordList : [];
|
|
48
60
|
|
|
49
|
-
if (
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
.
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
.map((word) => word.word);
|
|
61
|
-
} else {
|
|
62
|
-
wordsToCheck = wordObjects.map((word) => word.word);
|
|
63
|
-
}
|
|
61
|
+
if (baseWordsToCheck.length === 0) {
|
|
62
|
+
const filteredWords = wordObjects.filter((word) => {
|
|
63
|
+
const matchCategory = categories
|
|
64
|
+
? categories.includes(word.category)
|
|
65
|
+
: true;
|
|
66
|
+
const matchRegion = regions ? regions.includes(word.region) : true;
|
|
67
|
+
const matchSeverity = word.severity >= severityThreshold;
|
|
68
|
+
return matchCategory && matchRegion && matchSeverity;
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
baseWordsToCheck = filteredWords.map((word) => word.word);
|
|
64
72
|
}
|
|
65
73
|
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
74
|
+
const aliasMap = new Map<string, string>();
|
|
75
|
+
wordObjects.forEach((wordObj) => {
|
|
76
|
+
if (wordObj.aliases && wordObj.aliases.length > 0) {
|
|
77
|
+
const matchCategory = categories
|
|
78
|
+
? categories.includes(wordObj.category)
|
|
79
|
+
: true;
|
|
80
|
+
const matchRegion = regions ? regions.includes(wordObj.region) : true;
|
|
81
|
+
const matchSeverity = wordObj.severity >= severityThreshold;
|
|
82
|
+
|
|
83
|
+
if (matchCategory && matchRegion && matchSeverity) {
|
|
84
|
+
wordObj.aliases.forEach((alias) => {
|
|
85
|
+
aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
|
|
86
|
+
});
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
const wordsToCheck = [
|
|
92
|
+
...baseWordsToCheck,
|
|
93
|
+
...Array.from(aliasMap.keys()),
|
|
94
|
+
].filter((word) => !whitelist.includes(word.toLowerCase()));
|
|
69
95
|
|
|
70
96
|
if (wordsToCheck.length === 0) {
|
|
71
97
|
return [];
|
|
72
98
|
}
|
|
73
99
|
|
|
74
100
|
const matches = new Set<string>();
|
|
101
|
+
const actualMatches = new Map<string, string[]>();
|
|
75
102
|
|
|
76
|
-
wordsToCheck
|
|
77
|
-
const regex = createWordRegex(word, {
|
|
78
|
-
wholeWord: !checkSubstring,
|
|
79
|
-
caseSensitive: false,
|
|
80
|
-
leetSpeak: false,
|
|
81
|
-
detectSplit: false,
|
|
82
|
-
indonesianVariation: false,
|
|
83
|
-
});
|
|
103
|
+
initializeAhoCorasick(wordsToCheck);
|
|
84
104
|
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
105
|
+
const basicMatches = globalAhoCorasick.searchUnique(normalizedText);
|
|
106
|
+
for (const match of basicMatches) {
|
|
107
|
+
const originalWord =
|
|
108
|
+
aliasMap.get(match.toLowerCase()) || match.toLowerCase();
|
|
109
|
+
matches.add(originalWord);
|
|
110
|
+
|
|
111
|
+
if (!actualMatches.has(originalWord)) {
|
|
112
|
+
actualMatches.set(originalWord, []);
|
|
88
113
|
}
|
|
89
|
-
|
|
114
|
+
actualMatches.get(originalWord)?.push(match);
|
|
115
|
+
}
|
|
90
116
|
|
|
91
117
|
if (detectLeetSpeak) {
|
|
92
118
|
wordsToCheck.forEach((word) => {
|
|
@@ -100,7 +126,14 @@ export function findProfanity(
|
|
|
100
126
|
|
|
101
127
|
let match;
|
|
102
128
|
while ((match = leetRegex.exec(text)) !== null) {
|
|
103
|
-
|
|
129
|
+
const originalWord =
|
|
130
|
+
aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
131
|
+
matches.add(originalWord);
|
|
132
|
+
|
|
133
|
+
if (!actualMatches.has(originalWord)) {
|
|
134
|
+
actualMatches.set(originalWord, []);
|
|
135
|
+
}
|
|
136
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
104
137
|
}
|
|
105
138
|
});
|
|
106
139
|
}
|
|
@@ -117,41 +150,85 @@ export function findProfanity(
|
|
|
117
150
|
|
|
118
151
|
let match;
|
|
119
152
|
while ((match = variantRegex.exec(text)) !== null) {
|
|
120
|
-
|
|
153
|
+
const originalWord =
|
|
154
|
+
aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
155
|
+
matches.add(originalWord);
|
|
156
|
+
|
|
157
|
+
if (!actualMatches.has(originalWord)) {
|
|
158
|
+
actualMatches.set(originalWord, []);
|
|
159
|
+
}
|
|
160
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
121
161
|
}
|
|
122
162
|
});
|
|
123
163
|
}
|
|
124
164
|
|
|
125
165
|
if (detectSplit) {
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
});
|
|
166
|
+
wordsToCheck.forEach((word) => {
|
|
167
|
+
const splitRegex = createWordRegex(word, {
|
|
168
|
+
wholeWord: false,
|
|
169
|
+
caseSensitive: false,
|
|
170
|
+
leetSpeak: false,
|
|
171
|
+
detectSplit: true,
|
|
172
|
+
indonesianVariation: false,
|
|
173
|
+
});
|
|
135
174
|
|
|
136
|
-
|
|
137
|
-
|
|
175
|
+
let match;
|
|
176
|
+
while ((match = splitRegex.exec(text)) !== null) {
|
|
177
|
+
const originalWord =
|
|
178
|
+
aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
179
|
+
matches.add(originalWord);
|
|
180
|
+
|
|
181
|
+
if (!actualMatches.has(originalWord)) {
|
|
182
|
+
actualMatches.set(originalWord, []);
|
|
138
183
|
}
|
|
139
|
-
|
|
140
|
-
|
|
184
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
185
|
+
}
|
|
186
|
+
});
|
|
141
187
|
}
|
|
142
188
|
|
|
143
189
|
if (detectSimilarity) {
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
190
|
+
if (useLevenshtein) {
|
|
191
|
+
const possibleProfanity = findProfanityByLevenshteinDistance(
|
|
192
|
+
text,
|
|
193
|
+
wordsToCheck,
|
|
194
|
+
similarityThreshold,
|
|
195
|
+
maxLevenshteinDistance,
|
|
196
|
+
);
|
|
197
|
+
|
|
198
|
+
possibleProfanity.forEach((item) => {
|
|
199
|
+
const originalWord =
|
|
200
|
+
aliasMap.get(item.original.toLowerCase()) ||
|
|
201
|
+
item.original.toLowerCase();
|
|
202
|
+
matches.add(originalWord);
|
|
203
|
+
|
|
204
|
+
if (!actualMatches.has(originalWord)) {
|
|
205
|
+
actualMatches.set(originalWord, []);
|
|
206
|
+
}
|
|
207
|
+
actualMatches.get(originalWord)?.push(item.word);
|
|
208
|
+
});
|
|
209
|
+
} else {
|
|
210
|
+
const possibleProfanity = findPossibleProfanityBySimiliarity(
|
|
211
|
+
text,
|
|
212
|
+
wordsToCheck,
|
|
213
|
+
similarityThreshold,
|
|
214
|
+
);
|
|
215
|
+
|
|
216
|
+
possibleProfanity.forEach((item) => {
|
|
217
|
+
matches.add(item.original.toLowerCase());
|
|
218
|
+
|
|
219
|
+
const originalWord =
|
|
220
|
+
aliasMap.get(item.original.toLowerCase()) ||
|
|
221
|
+
item.original.toLowerCase();
|
|
222
|
+
if (!actualMatches.has(originalWord)) {
|
|
223
|
+
actualMatches.set(originalWord, []);
|
|
224
|
+
}
|
|
225
|
+
actualMatches.get(originalWord)?.push(item.word);
|
|
226
|
+
});
|
|
227
|
+
}
|
|
153
228
|
}
|
|
154
229
|
|
|
230
|
+
(findProfanity as FindProfanityFunction).lastActualMatches = actualMatches;
|
|
231
|
+
|
|
155
232
|
return Array.from(matches);
|
|
156
233
|
}
|
|
157
234
|
|
package/src/index.ts
CHANGED
|
@@ -21,7 +21,6 @@ import {
|
|
|
21
21
|
CATEGORY_PRESETS,
|
|
22
22
|
REGION_PRESETS,
|
|
23
23
|
getPresetOptions,
|
|
24
|
-
makeRandomGrawlixString,
|
|
25
24
|
} from "./config/options";
|
|
26
25
|
|
|
27
26
|
export class IDProfanityFilter {
|
|
@@ -161,10 +160,30 @@ export class IDProfanityFilter {
|
|
|
161
160
|
/**
|
|
162
161
|
* Mengaktifkan deteksi berdasarkan kesamaan
|
|
163
162
|
* @param threshold Threshold kesamaan (0-1)
|
|
163
|
+
* @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
|
|
164
|
+
* @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
|
|
164
165
|
*/
|
|
165
|
-
enableSimilarityDetection(
|
|
166
|
+
enableSimilarityDetection(
|
|
167
|
+
threshold: number = 0.8,
|
|
168
|
+
useLevenshtein: boolean = false,
|
|
169
|
+
maxLevenshteinDistance: number = 2,
|
|
170
|
+
) {
|
|
171
|
+
this.options.detectSimilarity = true;
|
|
172
|
+
this.options.similarityThreshold = threshold;
|
|
173
|
+
this.options.useLevenshtein = useLevenshtein;
|
|
174
|
+
this.options.maxLevenshteinDistance = maxLevenshteinDistance;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* Mengaktifkan deteksi berbasis Levenshtein distance
|
|
179
|
+
* @param threshold Threshold kesamaan (0-1)
|
|
180
|
+
* @param maxDistance Jarak maksimal Levenshtein (default: 2)
|
|
181
|
+
*/
|
|
182
|
+
enableLevenshteinDetection(threshold: number = 0.8, maxDistance: number = 2) {
|
|
166
183
|
this.options.detectSimilarity = true;
|
|
184
|
+
this.options.useLevenshtein = true;
|
|
167
185
|
this.options.similarityThreshold = threshold;
|
|
186
|
+
this.options.maxLevenshteinDistance = maxDistance;
|
|
168
187
|
}
|
|
169
188
|
}
|
|
170
189
|
|
package/src/types/index.ts
CHANGED
|
@@ -50,9 +50,11 @@ export interface FilterOptions {
|
|
|
50
50
|
useRandomGrawlix?: boolean;
|
|
51
51
|
keepFirstAndLast?: boolean;
|
|
52
52
|
indonesianVariation?: boolean;
|
|
53
|
-
detectSimilarity?: boolean;
|
|
54
|
-
similarityThreshold?: number;
|
|
53
|
+
detectSimilarity?: boolean; // Enable/disable Levenshtein distance matching
|
|
54
|
+
similarityThreshold?: number; // Threshold for Levenshtein distance similarity (0-1)
|
|
55
55
|
detectSplit?: boolean;
|
|
56
|
+
useLevenshtein?: boolean; // New option specifically for Levenshtein algorithm
|
|
57
|
+
maxLevenshteinDistance?: number; // Maximum allowed Levenshtein distance
|
|
56
58
|
}
|
|
57
59
|
|
|
58
60
|
export interface FilterResult {
|