@sideid/id-profanity-filter 1.9.6 → 1.10.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.eslintrc.js +30 -2
- package/README.md +45 -3
- package/dist/config/options.d.ts +24 -0
- package/dist/constants/wordList.d.ts +1 -1
- package/dist/core/analyzer.d.ts +1 -1
- package/dist/core/filter.d.ts +1 -1
- package/dist/core/matcher.d.ts +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.esm.js +411 -77
- package/dist/index.esm.js.map +1 -1
- package/dist/index.js +412 -76
- package/dist/index.js.map +1 -1
- package/dist/types/index.d.ts +2 -0
- package/dist/utils/similarityUtils.d.ts +25 -0
- package/eslint.config.mjs +40 -0
- package/package.json +2 -2
- package/src/config/options.ts +2 -0
- package/src/constants/regions/general.ts +102 -2
- package/src/constants/regions/jawa.ts +10 -0
- package/src/constants/wordList.ts +15 -8
- package/src/core/analyzer.ts +1 -7
- package/src/core/filter.ts +178 -37
- package/src/core/matcher.ts +128 -58
- package/src/index.ts +21 -2
- package/src/types/index.ts +4 -2
- package/src/utils/regexUtils.ts +0 -1
- package/src/utils/similarityUtils.ts +97 -2
- package/test.js +184 -0
- package/src/constants/categories/index.ts +0 -31
- package/src/constants/regions/index.ts +0 -62
package/dist/index.js
CHANGED
|
@@ -8,7 +8,7 @@ const general = [
|
|
|
8
8
|
category: "profanity",
|
|
9
9
|
region: "general",
|
|
10
10
|
severity: 0.7,
|
|
11
|
-
aliases: ["anjay", "anjir", "anying", "njing", "anj"],
|
|
11
|
+
aliases: ["anjay", "anjir", "anying", "njing", "anj", "anjg", "ajg"],
|
|
12
12
|
description: "Mengacu pada hewan anjing, digunakan sebagai umpatan",
|
|
13
13
|
context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
|
|
14
14
|
},
|
|
@@ -17,7 +17,7 @@ const general = [
|
|
|
17
17
|
category: "profanity",
|
|
18
18
|
region: "general",
|
|
19
19
|
severity: 0.6,
|
|
20
|
-
aliases: ["bab1", "b4b1"],
|
|
20
|
+
aliases: ["bab1", "b4b1", "b4bi", "8481", "8ab1", "ba81"],
|
|
21
21
|
description: "Mengacu pada hewan babi, digunakan sebagai umpatan",
|
|
22
22
|
context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
|
|
23
23
|
},
|
|
@@ -102,6 +102,105 @@ const general = [
|
|
|
102
102
|
description: "Kata yang mengacu pada orang yang banyak bicara",
|
|
103
103
|
context: "Hinaan untuk menyebut orang yang banyak bicara atau cerewet",
|
|
104
104
|
},
|
|
105
|
+
{
|
|
106
|
+
word: "ngentot",
|
|
107
|
+
category: "sexual",
|
|
108
|
+
region: "general",
|
|
109
|
+
severity: 0.9,
|
|
110
|
+
aliases: ["ngentod", "ntot", "tod"],
|
|
111
|
+
description: "Istilah kasar untuk aktivitas seksual",
|
|
112
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual",
|
|
113
|
+
},
|
|
114
|
+
{
|
|
115
|
+
word: "sialan",
|
|
116
|
+
category: "insult",
|
|
117
|
+
region: "general",
|
|
118
|
+
severity: 0.5,
|
|
119
|
+
aliases: ["sialn", "sl"],
|
|
120
|
+
description: "Kata yang mengacu pada orang yang membawa sial",
|
|
121
|
+
context: "Hinaan untuk menyebut orang yang dianggap membawa sial",
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
word: "pler",
|
|
125
|
+
category: "sexual",
|
|
126
|
+
region: "general",
|
|
127
|
+
severity: 0.9,
|
|
128
|
+
aliases: ["peler", "plr", "biji"],
|
|
129
|
+
description: "Istilah kasar untuk alat kelamin laki-laki",
|
|
130
|
+
context: "Kata vulgar yang merujuk pada alat kelamin laki-laki",
|
|
131
|
+
},
|
|
132
|
+
{
|
|
133
|
+
word: "bokep",
|
|
134
|
+
category: "sexual",
|
|
135
|
+
region: "general",
|
|
136
|
+
severity: 0.7,
|
|
137
|
+
aliases: ["bkp", "bokap"],
|
|
138
|
+
description: "Istilah untuk video atau konten pornografi",
|
|
139
|
+
context: "Kata yang mengacu pada materi pornografi",
|
|
140
|
+
},
|
|
141
|
+
{
|
|
142
|
+
word: "coli",
|
|
143
|
+
category: "sexual",
|
|
144
|
+
region: "general",
|
|
145
|
+
severity: 0.8,
|
|
146
|
+
aliases: ["col", "coly"],
|
|
147
|
+
description: "Istilah untuk masturbasi laki-laki",
|
|
148
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual pribadi",
|
|
149
|
+
},
|
|
150
|
+
{
|
|
151
|
+
word: "desah",
|
|
152
|
+
category: "sexual",
|
|
153
|
+
region: "general",
|
|
154
|
+
severity: 0.6,
|
|
155
|
+
aliases: ["ds4h", "dsh"],
|
|
156
|
+
description: "Istilah untuk suara yang dibuat selama aktivitas seksual",
|
|
157
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
158
|
+
},
|
|
159
|
+
{
|
|
160
|
+
word: "seks",
|
|
161
|
+
category: "sexual",
|
|
162
|
+
region: "general",
|
|
163
|
+
severity: 0.5,
|
|
164
|
+
aliases: ["sex", "ML"],
|
|
165
|
+
description: "Istilah untuk aktivitas seksual",
|
|
166
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
167
|
+
},
|
|
168
|
+
{
|
|
169
|
+
word: "kondom",
|
|
170
|
+
category: "sexual",
|
|
171
|
+
region: "general",
|
|
172
|
+
severity: 0.5,
|
|
173
|
+
aliases: ["kndm", "kondom", "cd"],
|
|
174
|
+
description: "Alat kontrasepsi",
|
|
175
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
176
|
+
},
|
|
177
|
+
{
|
|
178
|
+
word: "ngewe",
|
|
179
|
+
category: "sexual",
|
|
180
|
+
region: "general",
|
|
181
|
+
severity: 0.9,
|
|
182
|
+
aliases: ["ngew", "we"],
|
|
183
|
+
description: "Istilah kasar untuk aktivitas seksual",
|
|
184
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual",
|
|
185
|
+
},
|
|
186
|
+
{
|
|
187
|
+
word: "puki",
|
|
188
|
+
category: "sexual",
|
|
189
|
+
region: "general",
|
|
190
|
+
severity: 0.9,
|
|
191
|
+
aliases: ["puk", "pukih"],
|
|
192
|
+
description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
|
|
193
|
+
context: "Kata vulgar yang merujuk pada anatomi seksual",
|
|
194
|
+
},
|
|
195
|
+
{
|
|
196
|
+
word: "xxx",
|
|
197
|
+
category: "sexual",
|
|
198
|
+
region: "general",
|
|
199
|
+
severity: 0.6,
|
|
200
|
+
aliases: ["xXx", "triplex"],
|
|
201
|
+
description: "Simbol yang sering digunakan untuk menandai konten pornografi",
|
|
202
|
+
context: "Digunakan untuk menandai konten seksual eksplisit",
|
|
203
|
+
},
|
|
105
204
|
];
|
|
106
205
|
general.map((item) => item.word);
|
|
107
206
|
|
|
@@ -196,6 +295,15 @@ const jawa = [
|
|
|
196
295
|
description: "Secara harfiah berarti kepalamu, digunakan sebagai umpatan ringan",
|
|
197
296
|
context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
|
|
198
297
|
},
|
|
298
|
+
{
|
|
299
|
+
word: "itil",
|
|
300
|
+
category: "sexual",
|
|
301
|
+
region: "jawa",
|
|
302
|
+
severity: 0.9,
|
|
303
|
+
aliases: ["itl", "itul"],
|
|
304
|
+
description: "Kata vulgar yang mengacu pada bagian dari alat kelamin perempuan",
|
|
305
|
+
context: "Kata vulgar yang merujuk pada anatomi seksual",
|
|
306
|
+
},
|
|
199
307
|
];
|
|
200
308
|
jawa.map((item) => item.word);
|
|
201
309
|
|
|
@@ -716,7 +824,6 @@ function addLeetSpeakVariations(pattern) {
|
|
|
716
824
|
t: ["t", "7", "+"],
|
|
717
825
|
z: ["z", "2"],
|
|
718
826
|
};
|
|
719
|
-
// Ganti tiap karakter dengan variasinya dalam grup character class
|
|
720
827
|
return pattern
|
|
721
828
|
.split("")
|
|
722
829
|
.map((char) => {
|
|
@@ -900,6 +1007,37 @@ function findMostSimilar(target, candidates, threshold = 0.7) {
|
|
|
900
1007
|
}
|
|
901
1008
|
return mostSimilar;
|
|
902
1009
|
}
|
|
1010
|
+
/**
|
|
1011
|
+
* Mencari string yang paling mirip dari array menggunakan Levenshtein distance
|
|
1012
|
+
*
|
|
1013
|
+
* @param target String target
|
|
1014
|
+
* @param candidates Array string kandidat
|
|
1015
|
+
* @param threshold Minimum kesamaan yang diterima (0-1)
|
|
1016
|
+
* @param maxDistance Jarak Levenshtein maksimal yang diterima (default: 3)
|
|
1017
|
+
* @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
|
|
1018
|
+
*/
|
|
1019
|
+
function findMostSimilarWithLevenshtein(target, candidates, threshold = 0.7, maxDistance = 3) {
|
|
1020
|
+
if (!candidates.length)
|
|
1021
|
+
return null;
|
|
1022
|
+
let maxSimilarity = 0;
|
|
1023
|
+
let minDistance = Infinity;
|
|
1024
|
+
let mostSimilar = null;
|
|
1025
|
+
for (const candidate of candidates) {
|
|
1026
|
+
if (Math.abs(target.length - candidate.length) > maxDistance)
|
|
1027
|
+
continue;
|
|
1028
|
+
const distance = levenshteinDistance(target, candidate);
|
|
1029
|
+
const similarity = stringSimilarity(target, candidate);
|
|
1030
|
+
if ((similarity > maxSimilarity && similarity >= threshold) ||
|
|
1031
|
+
(similarity >= threshold && distance < minDistance)) {
|
|
1032
|
+
maxSimilarity = similarity;
|
|
1033
|
+
minDistance = distance;
|
|
1034
|
+
mostSimilar = candidate;
|
|
1035
|
+
if (distance <= 1 || similarity > 0.95)
|
|
1036
|
+
break;
|
|
1037
|
+
}
|
|
1038
|
+
}
|
|
1039
|
+
return mostSimilar;
|
|
1040
|
+
}
|
|
903
1041
|
/**
|
|
904
1042
|
* Cek apakah string mungkin merupakan variasi dari kata kotor
|
|
905
1043
|
* menggunakan kesamaan string
|
|
@@ -958,10 +1096,8 @@ function clusterSimilarWords(words, threshold = 0.8) {
|
|
|
958
1096
|
*/
|
|
959
1097
|
function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.8) {
|
|
960
1098
|
const result = [];
|
|
961
|
-
// Pisahkan teks menjadi kata-kata
|
|
962
1099
|
const words = text.toLowerCase().split(/\s+/);
|
|
963
1100
|
for (const word of words) {
|
|
964
|
-
// Lewati kata-kata yang terlalu pendek
|
|
965
1101
|
if (word.length < 3)
|
|
966
1102
|
continue;
|
|
967
1103
|
for (const profanity of profanityWords) {
|
|
@@ -978,6 +1114,41 @@ function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.
|
|
|
978
1114
|
}
|
|
979
1115
|
return result;
|
|
980
1116
|
}
|
|
1117
|
+
/**
|
|
1118
|
+
* Cari kata-kata kotor yang mungkin dari teks menggunakan Levenshtein distance
|
|
1119
|
+
*
|
|
1120
|
+
* @param text Teks yang akan diperiksa
|
|
1121
|
+
* @param profanityWords Daftar kata kotor
|
|
1122
|
+
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
1123
|
+
* @param maxDistance Jarak Levenshtein maksimal (default: 2)
|
|
1124
|
+
* @returns Array kata yang mungkin merupakan kata kotor
|
|
1125
|
+
*/
|
|
1126
|
+
function findProfanityByLevenshteinDistance(text, profanityWords, threshold = 0.8, maxDistance = 2) {
|
|
1127
|
+
const result = [];
|
|
1128
|
+
const words = text.toLowerCase().split(/\s+/);
|
|
1129
|
+
for (const word of words) {
|
|
1130
|
+
if (word.length < 3)
|
|
1131
|
+
continue;
|
|
1132
|
+
for (const profanity of profanityWords) {
|
|
1133
|
+
if (Math.abs(word.length - profanity.length) > maxDistance)
|
|
1134
|
+
continue;
|
|
1135
|
+
const distance = levenshteinDistance(word, profanity);
|
|
1136
|
+
if (distance <= maxDistance) {
|
|
1137
|
+
const similarity = stringSimilarity(word, profanity);
|
|
1138
|
+
if (similarity >= threshold) {
|
|
1139
|
+
result.push({
|
|
1140
|
+
word,
|
|
1141
|
+
original: profanity,
|
|
1142
|
+
similarity,
|
|
1143
|
+
distance,
|
|
1144
|
+
});
|
|
1145
|
+
break;
|
|
1146
|
+
}
|
|
1147
|
+
}
|
|
1148
|
+
}
|
|
1149
|
+
}
|
|
1150
|
+
return result;
|
|
1151
|
+
}
|
|
981
1152
|
|
|
982
1153
|
const DEFAULT_OPTIONS = {
|
|
983
1154
|
replaceWith: "*",
|
|
@@ -986,6 +1157,8 @@ const DEFAULT_OPTIONS = {
|
|
|
986
1157
|
checkSubstring: false,
|
|
987
1158
|
whitelist: [],
|
|
988
1159
|
severityThreshold: 0,
|
|
1160
|
+
useLevenshtein: false,
|
|
1161
|
+
maxLevenshteinDistance: 2,
|
|
989
1162
|
};
|
|
990
1163
|
const FILTER_PRESETS = {
|
|
991
1164
|
strict: {
|
|
@@ -1131,31 +1304,44 @@ function makeRandomGrawlixString(length) {
|
|
|
1131
1304
|
}
|
|
1132
1305
|
|
|
1133
1306
|
function findProfanity(text, options = {}) {
|
|
1134
|
-
const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, } = { ...DEFAULT_OPTIONS, ...options };
|
|
1307
|
+
const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, useLevenshtein = false, maxLevenshteinDistance = 2, } = { ...DEFAULT_OPTIONS, ...options };
|
|
1135
1308
|
const normalizedText = normalizeText(text);
|
|
1136
|
-
let
|
|
1137
|
-
if (
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
.
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
})
|
|
1148
|
-
.map((word) => word.word);
|
|
1149
|
-
}
|
|
1150
|
-
else {
|
|
1151
|
-
wordsToCheck = wordObjects.map((word) => word.word);
|
|
1152
|
-
}
|
|
1309
|
+
let baseWordsToCheck = wordList.length > 0 ? wordList : [];
|
|
1310
|
+
if (baseWordsToCheck.length === 0) {
|
|
1311
|
+
const filteredWords = wordObjects.filter((word) => {
|
|
1312
|
+
const matchCategory = categories
|
|
1313
|
+
? categories.includes(word.category)
|
|
1314
|
+
: true;
|
|
1315
|
+
const matchRegion = regions ? regions.includes(word.region) : true;
|
|
1316
|
+
const matchSeverity = word.severity >= severityThreshold;
|
|
1317
|
+
return matchCategory && matchRegion && matchSeverity;
|
|
1318
|
+
});
|
|
1319
|
+
baseWordsToCheck = filteredWords.map((word) => word.word);
|
|
1153
1320
|
}
|
|
1154
|
-
|
|
1321
|
+
const aliasMap = new Map();
|
|
1322
|
+
wordObjects.forEach((wordObj) => {
|
|
1323
|
+
if (wordObj.aliases && wordObj.aliases.length > 0) {
|
|
1324
|
+
const matchCategory = categories
|
|
1325
|
+
? categories.includes(wordObj.category)
|
|
1326
|
+
: true;
|
|
1327
|
+
const matchRegion = regions ? regions.includes(wordObj.region) : true;
|
|
1328
|
+
const matchSeverity = wordObj.severity >= severityThreshold;
|
|
1329
|
+
if (matchCategory && matchRegion && matchSeverity) {
|
|
1330
|
+
wordObj.aliases.forEach((alias) => {
|
|
1331
|
+
aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
|
|
1332
|
+
});
|
|
1333
|
+
}
|
|
1334
|
+
}
|
|
1335
|
+
});
|
|
1336
|
+
const wordsToCheck = [
|
|
1337
|
+
...baseWordsToCheck,
|
|
1338
|
+
...Array.from(aliasMap.keys()),
|
|
1339
|
+
].filter((word) => !whitelist.includes(word.toLowerCase()));
|
|
1155
1340
|
if (wordsToCheck.length === 0) {
|
|
1156
1341
|
return [];
|
|
1157
1342
|
}
|
|
1158
1343
|
const matches = new Set();
|
|
1344
|
+
const actualMatches = new Map();
|
|
1159
1345
|
wordsToCheck.forEach((word) => {
|
|
1160
1346
|
const regex = createWordRegex(word, {
|
|
1161
1347
|
wholeWord: !checkSubstring,
|
|
@@ -1164,8 +1350,14 @@ function findProfanity(text, options = {}) {
|
|
|
1164
1350
|
detectSplit: false,
|
|
1165
1351
|
indonesianVariation: false,
|
|
1166
1352
|
});
|
|
1167
|
-
|
|
1168
|
-
|
|
1353
|
+
let match;
|
|
1354
|
+
while ((match = regex.exec(normalizedText)) !== null) {
|
|
1355
|
+
const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
1356
|
+
matches.add(originalWord);
|
|
1357
|
+
if (!actualMatches.has(originalWord)) {
|
|
1358
|
+
actualMatches.set(originalWord, []);
|
|
1359
|
+
}
|
|
1360
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
1169
1361
|
}
|
|
1170
1362
|
});
|
|
1171
1363
|
if (detectLeetSpeak) {
|
|
@@ -1177,8 +1369,14 @@ function findProfanity(text, options = {}) {
|
|
|
1177
1369
|
detectSplit: false,
|
|
1178
1370
|
indonesianVariation: false,
|
|
1179
1371
|
});
|
|
1180
|
-
|
|
1181
|
-
|
|
1372
|
+
let match;
|
|
1373
|
+
while ((match = leetRegex.exec(text)) !== null) {
|
|
1374
|
+
const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
1375
|
+
matches.add(originalWord);
|
|
1376
|
+
if (!actualMatches.has(originalWord)) {
|
|
1377
|
+
actualMatches.set(originalWord, []);
|
|
1378
|
+
}
|
|
1379
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
1182
1380
|
}
|
|
1183
1381
|
});
|
|
1184
1382
|
}
|
|
@@ -1191,33 +1389,64 @@ function findProfanity(text, options = {}) {
|
|
|
1191
1389
|
detectSplit: false,
|
|
1192
1390
|
indonesianVariation: true,
|
|
1193
1391
|
});
|
|
1194
|
-
|
|
1195
|
-
|
|
1392
|
+
let match;
|
|
1393
|
+
while ((match = variantRegex.exec(text)) !== null) {
|
|
1394
|
+
const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
1395
|
+
matches.add(originalWord);
|
|
1396
|
+
if (!actualMatches.has(originalWord)) {
|
|
1397
|
+
actualMatches.set(originalWord, []);
|
|
1398
|
+
}
|
|
1399
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
1196
1400
|
}
|
|
1197
1401
|
});
|
|
1198
1402
|
}
|
|
1199
1403
|
if (detectSplit) {
|
|
1200
|
-
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
|
|
1206
|
-
|
|
1207
|
-
indonesianVariation: false,
|
|
1208
|
-
});
|
|
1209
|
-
if (splitRegex.test(text)) {
|
|
1210
|
-
matches.add(word.toLowerCase());
|
|
1211
|
-
}
|
|
1404
|
+
wordsToCheck.forEach((word) => {
|
|
1405
|
+
const splitRegex = createWordRegex(word, {
|
|
1406
|
+
wholeWord: false,
|
|
1407
|
+
caseSensitive: false,
|
|
1408
|
+
leetSpeak: false,
|
|
1409
|
+
detectSplit: true,
|
|
1410
|
+
indonesianVariation: false,
|
|
1212
1411
|
});
|
|
1213
|
-
|
|
1412
|
+
let match;
|
|
1413
|
+
while ((match = splitRegex.exec(text)) !== null) {
|
|
1414
|
+
const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
1415
|
+
matches.add(originalWord);
|
|
1416
|
+
if (!actualMatches.has(originalWord)) {
|
|
1417
|
+
actualMatches.set(originalWord, []);
|
|
1418
|
+
}
|
|
1419
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
1420
|
+
}
|
|
1421
|
+
});
|
|
1214
1422
|
}
|
|
1215
1423
|
if (detectSimilarity) {
|
|
1216
|
-
|
|
1217
|
-
|
|
1218
|
-
|
|
1219
|
-
|
|
1424
|
+
if (useLevenshtein) {
|
|
1425
|
+
const possibleProfanity = findProfanityByLevenshteinDistance(text, wordsToCheck, similarityThreshold, maxLevenshteinDistance);
|
|
1426
|
+
possibleProfanity.forEach((item) => {
|
|
1427
|
+
const originalWord = aliasMap.get(item.original.toLowerCase()) ||
|
|
1428
|
+
item.original.toLowerCase();
|
|
1429
|
+
matches.add(originalWord);
|
|
1430
|
+
if (!actualMatches.has(originalWord)) {
|
|
1431
|
+
actualMatches.set(originalWord, []);
|
|
1432
|
+
}
|
|
1433
|
+
actualMatches.get(originalWord)?.push(item.word);
|
|
1434
|
+
});
|
|
1435
|
+
}
|
|
1436
|
+
else {
|
|
1437
|
+
const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
|
|
1438
|
+
possibleProfanity.forEach((item) => {
|
|
1439
|
+
matches.add(item.original.toLowerCase());
|
|
1440
|
+
const originalWord = aliasMap.get(item.original.toLowerCase()) ||
|
|
1441
|
+
item.original.toLowerCase();
|
|
1442
|
+
if (!actualMatches.has(originalWord)) {
|
|
1443
|
+
actualMatches.set(originalWord, []);
|
|
1444
|
+
}
|
|
1445
|
+
actualMatches.get(originalWord)?.push(item.word);
|
|
1446
|
+
});
|
|
1447
|
+
}
|
|
1220
1448
|
}
|
|
1449
|
+
findProfanity.lastActualMatches = actualMatches;
|
|
1221
1450
|
return Array.from(matches);
|
|
1222
1451
|
}
|
|
1223
1452
|
/**
|
|
@@ -1307,14 +1536,20 @@ function calculateSeverity(matchDetails) {
|
|
|
1307
1536
|
* @returns FilterResult dengan hasil filter
|
|
1308
1537
|
*/
|
|
1309
1538
|
function filter(text, options = {}) {
|
|
1310
|
-
const { replaceWith =
|
|
1539
|
+
const { replaceWith = '*', fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, detectSplit = false, detectSimilarity = false, useLevenshtein = false, maxLevenshteinDistance = 2, similarityThreshold = 0.8, } = { ...DEFAULT_OPTIONS, ...options };
|
|
1311
1540
|
const matches = findProfanity(text, {
|
|
1312
1541
|
...options,
|
|
1313
1542
|
detectLeetSpeak,
|
|
1314
1543
|
whitelist,
|
|
1315
1544
|
checkSubstring,
|
|
1316
1545
|
indonesianVariation,
|
|
1546
|
+
detectSplit,
|
|
1547
|
+
detectSimilarity,
|
|
1548
|
+
useLevenshtein,
|
|
1549
|
+
maxLevenshteinDistance,
|
|
1550
|
+
similarityThreshold,
|
|
1317
1551
|
});
|
|
1552
|
+
const actualMatches = findProfanity.lastActualMatches || new Map();
|
|
1318
1553
|
const matchDetails = findProfanityWithMetadata(text, options);
|
|
1319
1554
|
if (matches.length === 0) {
|
|
1320
1555
|
return {
|
|
@@ -1329,36 +1564,120 @@ function filter(text, options = {}) {
|
|
|
1329
1564
|
const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
|
|
1330
1565
|
(m.aliases &&
|
|
1331
1566
|
m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
|
|
1332
|
-
const
|
|
1333
|
-
|
|
1334
|
-
|
|
1335
|
-
|
|
1336
|
-
|
|
1337
|
-
|
|
1567
|
+
const variants = actualMatches.get(word.toLowerCase()) || [];
|
|
1568
|
+
variants.push(word);
|
|
1569
|
+
const uniqueVariants = [...new Set(variants)];
|
|
1570
|
+
uniqueVariants.forEach((variant) => {
|
|
1571
|
+
const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, 'gi');
|
|
1572
|
+
let match;
|
|
1573
|
+
while ((match = regex.exec(filteredText)) !== null) {
|
|
1574
|
+
const originalWord = match[0];
|
|
1575
|
+
if (whitelist.includes(originalWord.toLowerCase()))
|
|
1576
|
+
continue;
|
|
1577
|
+
let censoredWord;
|
|
1578
|
+
if (useRandomGrawlix) {
|
|
1579
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
1580
|
+
}
|
|
1581
|
+
else {
|
|
1582
|
+
censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
|
|
1583
|
+
}
|
|
1584
|
+
replacements.push({
|
|
1585
|
+
original: originalWord,
|
|
1586
|
+
censored: censoredWord,
|
|
1587
|
+
metadata,
|
|
1588
|
+
});
|
|
1589
|
+
filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'), censoredWord);
|
|
1590
|
+
}
|
|
1338
1591
|
});
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
|
|
1342
|
-
|
|
1343
|
-
|
|
1344
|
-
|
|
1345
|
-
|
|
1346
|
-
|
|
1347
|
-
|
|
1348
|
-
|
|
1592
|
+
if (detectSplit || detectLeetSpeak) {
|
|
1593
|
+
if (detectLeetSpeak) {
|
|
1594
|
+
const leetRegex = createWordRegex(word, {
|
|
1595
|
+
wholeWord: true,
|
|
1596
|
+
caseSensitive: false,
|
|
1597
|
+
leetSpeak: true,
|
|
1598
|
+
detectSplit: false,
|
|
1599
|
+
indonesianVariation: false,
|
|
1600
|
+
});
|
|
1601
|
+
let match;
|
|
1602
|
+
while ((match = leetRegex.exec(filteredText)) !== null) {
|
|
1603
|
+
const originalWord = match[0];
|
|
1604
|
+
if (whitelist.includes(originalWord.toLowerCase()))
|
|
1605
|
+
continue;
|
|
1606
|
+
let censoredWord;
|
|
1607
|
+
if (useRandomGrawlix) {
|
|
1608
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
1609
|
+
}
|
|
1610
|
+
else {
|
|
1611
|
+
censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
|
|
1612
|
+
}
|
|
1613
|
+
replacements.push({
|
|
1614
|
+
original: originalWord,
|
|
1615
|
+
censored: censoredWord,
|
|
1616
|
+
metadata,
|
|
1617
|
+
});
|
|
1618
|
+
filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), 'g'), censoredWord);
|
|
1619
|
+
}
|
|
1349
1620
|
}
|
|
1350
|
-
|
|
1351
|
-
|
|
1621
|
+
if (detectSplit) {
|
|
1622
|
+
const splitRegex = createWordRegex(word, {
|
|
1623
|
+
wholeWord: false,
|
|
1624
|
+
caseSensitive: false,
|
|
1625
|
+
leetSpeak: false,
|
|
1626
|
+
detectSplit: true,
|
|
1627
|
+
indonesianVariation: false,
|
|
1628
|
+
});
|
|
1629
|
+
let match;
|
|
1630
|
+
while ((match = splitRegex.exec(filteredText)) !== null) {
|
|
1631
|
+
const originalWord = match[0];
|
|
1632
|
+
if (whitelist.includes(originalWord.toLowerCase()))
|
|
1633
|
+
continue;
|
|
1634
|
+
let censoredWord;
|
|
1635
|
+
if (useRandomGrawlix) {
|
|
1636
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
1637
|
+
}
|
|
1638
|
+
else {
|
|
1639
|
+
censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
|
|
1640
|
+
}
|
|
1641
|
+
replacements.push({
|
|
1642
|
+
original: originalWord,
|
|
1643
|
+
censored: censoredWord,
|
|
1644
|
+
metadata,
|
|
1645
|
+
});
|
|
1646
|
+
filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), 'g'), censoredWord);
|
|
1647
|
+
}
|
|
1352
1648
|
}
|
|
1353
|
-
replacements.push({
|
|
1354
|
-
original: originalWord,
|
|
1355
|
-
censored: censoredWord,
|
|
1356
|
-
metadata,
|
|
1357
|
-
});
|
|
1358
|
-
const replaceRegex = new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g");
|
|
1359
|
-
filteredText = filteredText.replace(replaceRegex, censoredWord);
|
|
1360
1649
|
}
|
|
1361
1650
|
});
|
|
1651
|
+
if (detectSimilarity && useLevenshtein) {
|
|
1652
|
+
matches.forEach((word) => {
|
|
1653
|
+
const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
|
|
1654
|
+
(m.aliases &&
|
|
1655
|
+
m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
|
|
1656
|
+
const variants = actualMatches.get(word.toLowerCase()) || [];
|
|
1657
|
+
variants.forEach((variant) => {
|
|
1658
|
+
const exactVariantRegex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, 'gi');
|
|
1659
|
+
let match;
|
|
1660
|
+
while ((match = exactVariantRegex.exec(filteredText)) !== null) {
|
|
1661
|
+
const originalWord = match[0];
|
|
1662
|
+
if (whitelist.includes(originalWord.toLowerCase()))
|
|
1663
|
+
continue;
|
|
1664
|
+
let censoredWord;
|
|
1665
|
+
if (useRandomGrawlix) {
|
|
1666
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
1667
|
+
}
|
|
1668
|
+
else {
|
|
1669
|
+
censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
|
|
1670
|
+
}
|
|
1671
|
+
replacements.push({
|
|
1672
|
+
original: originalWord,
|
|
1673
|
+
censored: censoredWord,
|
|
1674
|
+
metadata,
|
|
1675
|
+
});
|
|
1676
|
+
filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'), censoredWord);
|
|
1677
|
+
}
|
|
1678
|
+
});
|
|
1679
|
+
});
|
|
1680
|
+
}
|
|
1362
1681
|
return {
|
|
1363
1682
|
filtered: filteredText,
|
|
1364
1683
|
censored: replacements.length,
|
|
@@ -1501,9 +1820,9 @@ function analyzeWithContext(text, contextWindowSize = 5, options = {}) {
|
|
|
1501
1820
|
const regex = createContextRegex(word, contextWindowSize);
|
|
1502
1821
|
let match;
|
|
1503
1822
|
while ((match = regex.exec(text)) !== null) {
|
|
1504
|
-
const beforeContext = match[1] ||
|
|
1823
|
+
const beforeContext = match[1] || '';
|
|
1505
1824
|
const wordMatch = match[2];
|
|
1506
|
-
const afterContext = match[3] ||
|
|
1825
|
+
const afterContext = match[3] || '';
|
|
1507
1826
|
result.push({
|
|
1508
1827
|
word: wordMatch,
|
|
1509
1828
|
context: beforeContext + wordMatch + afterContext,
|
|
@@ -1633,10 +1952,25 @@ class IDProfanityFilter {
|
|
|
1633
1952
|
/**
|
|
1634
1953
|
* Mengaktifkan deteksi berdasarkan kesamaan
|
|
1635
1954
|
* @param threshold Threshold kesamaan (0-1)
|
|
1955
|
+
* @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
|
|
1956
|
+
* @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
|
|
1957
|
+
*/
|
|
1958
|
+
enableSimilarityDetection(threshold = 0.8, useLevenshtein = false, maxLevenshteinDistance = 2) {
|
|
1959
|
+
this.options.detectSimilarity = true;
|
|
1960
|
+
this.options.similarityThreshold = threshold;
|
|
1961
|
+
this.options.useLevenshtein = useLevenshtein;
|
|
1962
|
+
this.options.maxLevenshteinDistance = maxLevenshteinDistance;
|
|
1963
|
+
}
|
|
1964
|
+
/**
|
|
1965
|
+
* Mengaktifkan deteksi berbasis Levenshtein distance
|
|
1966
|
+
* @param threshold Threshold kesamaan (0-1)
|
|
1967
|
+
* @param maxDistance Jarak maksimal Levenshtein (default: 2)
|
|
1636
1968
|
*/
|
|
1637
|
-
|
|
1969
|
+
enableLevenshteinDetection(threshold = 0.8, maxDistance = 2) {
|
|
1638
1970
|
this.options.detectSimilarity = true;
|
|
1971
|
+
this.options.useLevenshtein = true;
|
|
1639
1972
|
this.options.similarityThreshold = threshold;
|
|
1973
|
+
this.options.maxLevenshteinDistance = maxDistance;
|
|
1640
1974
|
}
|
|
1641
1975
|
}
|
|
1642
1976
|
const idFilter = {
|
|
@@ -1682,8 +2016,10 @@ exports.escapeRegExp = escapeRegExp;
|
|
|
1682
2016
|
exports.filter = filter;
|
|
1683
2017
|
exports.findCategories = findCategories;
|
|
1684
2018
|
exports.findMostSimilar = findMostSimilar;
|
|
2019
|
+
exports.findMostSimilarWithLevenshtein = findMostSimilarWithLevenshtein;
|
|
1685
2020
|
exports.findPossibleProfanityBySimiliarity = findPossibleProfanityBySimiliarity;
|
|
1686
2021
|
exports.findProfanity = findProfanity;
|
|
2022
|
+
exports.findProfanityByLevenshteinDistance = findProfanityByLevenshteinDistance;
|
|
1687
2023
|
exports.findProfanityWithMetadata = findProfanityWithMetadata;
|
|
1688
2024
|
exports.findRegions = findRegions;
|
|
1689
2025
|
exports.getContextAroundIndex = getContextAroundIndex;
|