@sideid/id-profanity-filter 1.9.6 → 1.10.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -8,7 +8,7 @@ const general = [
8
8
  category: "profanity",
9
9
  region: "general",
10
10
  severity: 0.7,
11
- aliases: ["anjay", "anjir", "anying", "njing", "anj"],
11
+ aliases: ["anjay", "anjir", "anying", "njing", "anj", "anjg", "ajg"],
12
12
  description: "Mengacu pada hewan anjing, digunakan sebagai umpatan",
13
13
  context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
14
14
  },
@@ -17,7 +17,7 @@ const general = [
17
17
  category: "profanity",
18
18
  region: "general",
19
19
  severity: 0.6,
20
- aliases: ["bab1", "b4b1"],
20
+ aliases: ["bab1", "b4b1", "b4bi", "8481", "8ab1", "ba81"],
21
21
  description: "Mengacu pada hewan babi, digunakan sebagai umpatan",
22
22
  context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
23
23
  },
@@ -102,6 +102,105 @@ const general = [
102
102
  description: "Kata yang mengacu pada orang yang banyak bicara",
103
103
  context: "Hinaan untuk menyebut orang yang banyak bicara atau cerewet",
104
104
  },
105
+ {
106
+ word: "ngentot",
107
+ category: "sexual",
108
+ region: "general",
109
+ severity: 0.9,
110
+ aliases: ["ngentod", "ntot", "tod"],
111
+ description: "Istilah kasar untuk aktivitas seksual",
112
+ context: "Kata vulgar yang merujuk pada aktivitas seksual",
113
+ },
114
+ {
115
+ word: "sialan",
116
+ category: "insult",
117
+ region: "general",
118
+ severity: 0.5,
119
+ aliases: ["sialn", "sl"],
120
+ description: "Kata yang mengacu pada orang yang membawa sial",
121
+ context: "Hinaan untuk menyebut orang yang dianggap membawa sial",
122
+ },
123
+ {
124
+ word: "pler",
125
+ category: "sexual",
126
+ region: "general",
127
+ severity: 0.9,
128
+ aliases: ["peler", "plr", "biji"],
129
+ description: "Istilah kasar untuk alat kelamin laki-laki",
130
+ context: "Kata vulgar yang merujuk pada alat kelamin laki-laki",
131
+ },
132
+ {
133
+ word: "bokep",
134
+ category: "sexual",
135
+ region: "general",
136
+ severity: 0.7,
137
+ aliases: ["bkp", "bokap"],
138
+ description: "Istilah untuk video atau konten pornografi",
139
+ context: "Kata yang mengacu pada materi pornografi",
140
+ },
141
+ {
142
+ word: "coli",
143
+ category: "sexual",
144
+ region: "general",
145
+ severity: 0.8,
146
+ aliases: ["col", "coly"],
147
+ description: "Istilah untuk masturbasi laki-laki",
148
+ context: "Kata vulgar yang merujuk pada aktivitas seksual pribadi",
149
+ },
150
+ {
151
+ word: "desah",
152
+ category: "sexual",
153
+ region: "general",
154
+ severity: 0.6,
155
+ aliases: ["ds4h", "dsh"],
156
+ description: "Istilah untuk suara yang dibuat selama aktivitas seksual",
157
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
158
+ },
159
+ {
160
+ word: "seks",
161
+ category: "sexual",
162
+ region: "general",
163
+ severity: 0.5,
164
+ aliases: ["sex", "ML"],
165
+ description: "Istilah untuk aktivitas seksual",
166
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
167
+ },
168
+ {
169
+ word: "kondom",
170
+ category: "sexual",
171
+ region: "general",
172
+ severity: 0.5,
173
+ aliases: ["kndm", "kondom", "cd"],
174
+ description: "Alat kontrasepsi",
175
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
176
+ },
177
+ {
178
+ word: "ngewe",
179
+ category: "sexual",
180
+ region: "general",
181
+ severity: 0.9,
182
+ aliases: ["ngew", "we"],
183
+ description: "Istilah kasar untuk aktivitas seksual",
184
+ context: "Kata vulgar yang merujuk pada aktivitas seksual",
185
+ },
186
+ {
187
+ word: "puki",
188
+ category: "sexual",
189
+ region: "general",
190
+ severity: 0.9,
191
+ aliases: ["puk", "pukih"],
192
+ description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
193
+ context: "Kata vulgar yang merujuk pada anatomi seksual",
194
+ },
195
+ {
196
+ word: "xxx",
197
+ category: "sexual",
198
+ region: "general",
199
+ severity: 0.6,
200
+ aliases: ["xXx", "triplex"],
201
+ description: "Simbol yang sering digunakan untuk menandai konten pornografi",
202
+ context: "Digunakan untuk menandai konten seksual eksplisit",
203
+ },
105
204
  ];
106
205
  general.map((item) => item.word);
107
206
 
@@ -196,6 +295,15 @@ const jawa = [
196
295
  description: "Secara harfiah berarti kepalamu, digunakan sebagai umpatan ringan",
197
296
  context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
198
297
  },
298
+ {
299
+ word: "itil",
300
+ category: "sexual",
301
+ region: "jawa",
302
+ severity: 0.9,
303
+ aliases: ["itl", "itul"],
304
+ description: "Kata vulgar yang mengacu pada bagian dari alat kelamin perempuan",
305
+ context: "Kata vulgar yang merujuk pada anatomi seksual",
306
+ },
199
307
  ];
200
308
  jawa.map((item) => item.word);
201
309
 
@@ -716,7 +824,6 @@ function addLeetSpeakVariations(pattern) {
716
824
  t: ["t", "7", "+"],
717
825
  z: ["z", "2"],
718
826
  };
719
- // Ganti tiap karakter dengan variasinya dalam grup character class
720
827
  return pattern
721
828
  .split("")
722
829
  .map((char) => {
@@ -900,6 +1007,37 @@ function findMostSimilar(target, candidates, threshold = 0.7) {
900
1007
  }
901
1008
  return mostSimilar;
902
1009
  }
1010
+ /**
1011
+ * Mencari string yang paling mirip dari array menggunakan Levenshtein distance
1012
+ *
1013
+ * @param target String target
1014
+ * @param candidates Array string kandidat
1015
+ * @param threshold Minimum kesamaan yang diterima (0-1)
1016
+ * @param maxDistance Jarak Levenshtein maksimal yang diterima (default: 3)
1017
+ * @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
1018
+ */
1019
+ function findMostSimilarWithLevenshtein(target, candidates, threshold = 0.7, maxDistance = 3) {
1020
+ if (!candidates.length)
1021
+ return null;
1022
+ let maxSimilarity = 0;
1023
+ let minDistance = Infinity;
1024
+ let mostSimilar = null;
1025
+ for (const candidate of candidates) {
1026
+ if (Math.abs(target.length - candidate.length) > maxDistance)
1027
+ continue;
1028
+ const distance = levenshteinDistance(target, candidate);
1029
+ const similarity = stringSimilarity(target, candidate);
1030
+ if ((similarity > maxSimilarity && similarity >= threshold) ||
1031
+ (similarity >= threshold && distance < minDistance)) {
1032
+ maxSimilarity = similarity;
1033
+ minDistance = distance;
1034
+ mostSimilar = candidate;
1035
+ if (distance <= 1 || similarity > 0.95)
1036
+ break;
1037
+ }
1038
+ }
1039
+ return mostSimilar;
1040
+ }
903
1041
  /**
904
1042
  * Cek apakah string mungkin merupakan variasi dari kata kotor
905
1043
  * menggunakan kesamaan string
@@ -958,10 +1096,8 @@ function clusterSimilarWords(words, threshold = 0.8) {
958
1096
  */
959
1097
  function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.8) {
960
1098
  const result = [];
961
- // Pisahkan teks menjadi kata-kata
962
1099
  const words = text.toLowerCase().split(/\s+/);
963
1100
  for (const word of words) {
964
- // Lewati kata-kata yang terlalu pendek
965
1101
  if (word.length < 3)
966
1102
  continue;
967
1103
  for (const profanity of profanityWords) {
@@ -978,6 +1114,41 @@ function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.
978
1114
  }
979
1115
  return result;
980
1116
  }
1117
+ /**
1118
+ * Cari kata-kata kotor yang mungkin dari teks menggunakan Levenshtein distance
1119
+ *
1120
+ * @param text Teks yang akan diperiksa
1121
+ * @param profanityWords Daftar kata kotor
1122
+ * @param threshold Batas minimum kesamaan (default: 0.8)
1123
+ * @param maxDistance Jarak Levenshtein maksimal (default: 2)
1124
+ * @returns Array kata yang mungkin merupakan kata kotor
1125
+ */
1126
+ function findProfanityByLevenshteinDistance(text, profanityWords, threshold = 0.8, maxDistance = 2) {
1127
+ const result = [];
1128
+ const words = text.toLowerCase().split(/\s+/);
1129
+ for (const word of words) {
1130
+ if (word.length < 3)
1131
+ continue;
1132
+ for (const profanity of profanityWords) {
1133
+ if (Math.abs(word.length - profanity.length) > maxDistance)
1134
+ continue;
1135
+ const distance = levenshteinDistance(word, profanity);
1136
+ if (distance <= maxDistance) {
1137
+ const similarity = stringSimilarity(word, profanity);
1138
+ if (similarity >= threshold) {
1139
+ result.push({
1140
+ word,
1141
+ original: profanity,
1142
+ similarity,
1143
+ distance,
1144
+ });
1145
+ break;
1146
+ }
1147
+ }
1148
+ }
1149
+ }
1150
+ return result;
1151
+ }
981
1152
 
982
1153
  const DEFAULT_OPTIONS = {
983
1154
  replaceWith: "*",
@@ -986,6 +1157,8 @@ const DEFAULT_OPTIONS = {
986
1157
  checkSubstring: false,
987
1158
  whitelist: [],
988
1159
  severityThreshold: 0,
1160
+ useLevenshtein: false,
1161
+ maxLevenshteinDistance: 2,
989
1162
  };
990
1163
  const FILTER_PRESETS = {
991
1164
  strict: {
@@ -1131,31 +1304,44 @@ function makeRandomGrawlixString(length) {
1131
1304
  }
1132
1305
 
1133
1306
  function findProfanity(text, options = {}) {
1134
- const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, } = { ...DEFAULT_OPTIONS, ...options };
1307
+ const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, useLevenshtein = false, maxLevenshteinDistance = 2, } = { ...DEFAULT_OPTIONS, ...options };
1135
1308
  const normalizedText = normalizeText(text);
1136
- let wordsToCheck = wordList.length > 0 ? wordList : [];
1137
- if (wordsToCheck.length === 0) {
1138
- if (categories || regions || severityThreshold > 0) {
1139
- wordsToCheck = wordObjects
1140
- .filter((word) => {
1141
- const matchCategory = categories
1142
- ? categories.includes(word.category)
1143
- : true;
1144
- const matchRegion = regions ? regions.includes(word.region) : true;
1145
- const matchSeverity = word.severity >= severityThreshold;
1146
- return matchCategory && matchRegion && matchSeverity;
1147
- })
1148
- .map((word) => word.word);
1149
- }
1150
- else {
1151
- wordsToCheck = wordObjects.map((word) => word.word);
1152
- }
1309
+ let baseWordsToCheck = wordList.length > 0 ? wordList : [];
1310
+ if (baseWordsToCheck.length === 0) {
1311
+ const filteredWords = wordObjects.filter((word) => {
1312
+ const matchCategory = categories
1313
+ ? categories.includes(word.category)
1314
+ : true;
1315
+ const matchRegion = regions ? regions.includes(word.region) : true;
1316
+ const matchSeverity = word.severity >= severityThreshold;
1317
+ return matchCategory && matchRegion && matchSeverity;
1318
+ });
1319
+ baseWordsToCheck = filteredWords.map((word) => word.word);
1153
1320
  }
1154
- wordsToCheck = wordsToCheck.filter((word) => !whitelist.includes(word.toLocaleLowerCase()));
1321
+ const aliasMap = new Map();
1322
+ wordObjects.forEach((wordObj) => {
1323
+ if (wordObj.aliases && wordObj.aliases.length > 0) {
1324
+ const matchCategory = categories
1325
+ ? categories.includes(wordObj.category)
1326
+ : true;
1327
+ const matchRegion = regions ? regions.includes(wordObj.region) : true;
1328
+ const matchSeverity = wordObj.severity >= severityThreshold;
1329
+ if (matchCategory && matchRegion && matchSeverity) {
1330
+ wordObj.aliases.forEach((alias) => {
1331
+ aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
1332
+ });
1333
+ }
1334
+ }
1335
+ });
1336
+ const wordsToCheck = [
1337
+ ...baseWordsToCheck,
1338
+ ...Array.from(aliasMap.keys()),
1339
+ ].filter((word) => !whitelist.includes(word.toLowerCase()));
1155
1340
  if (wordsToCheck.length === 0) {
1156
1341
  return [];
1157
1342
  }
1158
1343
  const matches = new Set();
1344
+ const actualMatches = new Map();
1159
1345
  wordsToCheck.forEach((word) => {
1160
1346
  const regex = createWordRegex(word, {
1161
1347
  wholeWord: !checkSubstring,
@@ -1164,8 +1350,14 @@ function findProfanity(text, options = {}) {
1164
1350
  detectSplit: false,
1165
1351
  indonesianVariation: false,
1166
1352
  });
1167
- while ((regex.exec(normalizedText)) !== null) {
1168
- matches.add(word.toLowerCase());
1353
+ let match;
1354
+ while ((match = regex.exec(normalizedText)) !== null) {
1355
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1356
+ matches.add(originalWord);
1357
+ if (!actualMatches.has(originalWord)) {
1358
+ actualMatches.set(originalWord, []);
1359
+ }
1360
+ actualMatches.get(originalWord)?.push(match[0]);
1169
1361
  }
1170
1362
  });
1171
1363
  if (detectLeetSpeak) {
@@ -1177,8 +1369,14 @@ function findProfanity(text, options = {}) {
1177
1369
  detectSplit: false,
1178
1370
  indonesianVariation: false,
1179
1371
  });
1180
- while ((leetRegex.exec(text)) !== null) {
1181
- matches.add(word.toLowerCase());
1372
+ let match;
1373
+ while ((match = leetRegex.exec(text)) !== null) {
1374
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1375
+ matches.add(originalWord);
1376
+ if (!actualMatches.has(originalWord)) {
1377
+ actualMatches.set(originalWord, []);
1378
+ }
1379
+ actualMatches.get(originalWord)?.push(match[0]);
1182
1380
  }
1183
1381
  });
1184
1382
  }
@@ -1191,33 +1389,64 @@ function findProfanity(text, options = {}) {
1191
1389
  detectSplit: false,
1192
1390
  indonesianVariation: true,
1193
1391
  });
1194
- while ((variantRegex.exec(text)) !== null) {
1195
- matches.add(word.toLowerCase());
1392
+ let match;
1393
+ while ((match = variantRegex.exec(text)) !== null) {
1394
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1395
+ matches.add(originalWord);
1396
+ if (!actualMatches.has(originalWord)) {
1397
+ actualMatches.set(originalWord, []);
1398
+ }
1399
+ actualMatches.get(originalWord)?.push(match[0]);
1196
1400
  }
1197
1401
  });
1198
1402
  }
1199
1403
  if (detectSplit) {
1200
- if (detectSplitWords(text, wordsToCheck)) {
1201
- wordsToCheck.forEach((word) => {
1202
- const splitRegex = createWordRegex(word, {
1203
- wholeWord: false,
1204
- caseSensitive: false,
1205
- leetSpeak: false,
1206
- detectSplit: true,
1207
- indonesianVariation: false,
1208
- });
1209
- if (splitRegex.test(text)) {
1210
- matches.add(word.toLowerCase());
1211
- }
1404
+ wordsToCheck.forEach((word) => {
1405
+ const splitRegex = createWordRegex(word, {
1406
+ wholeWord: false,
1407
+ caseSensitive: false,
1408
+ leetSpeak: false,
1409
+ detectSplit: true,
1410
+ indonesianVariation: false,
1212
1411
  });
1213
- }
1412
+ let match;
1413
+ while ((match = splitRegex.exec(text)) !== null) {
1414
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1415
+ matches.add(originalWord);
1416
+ if (!actualMatches.has(originalWord)) {
1417
+ actualMatches.set(originalWord, []);
1418
+ }
1419
+ actualMatches.get(originalWord)?.push(match[0]);
1420
+ }
1421
+ });
1214
1422
  }
1215
1423
  if (detectSimilarity) {
1216
- const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
1217
- possibleProfanity.forEach((item) => {
1218
- matches.add(item.original.toLowerCase());
1219
- });
1424
+ if (useLevenshtein) {
1425
+ const possibleProfanity = findProfanityByLevenshteinDistance(text, wordsToCheck, similarityThreshold, maxLevenshteinDistance);
1426
+ possibleProfanity.forEach((item) => {
1427
+ const originalWord = aliasMap.get(item.original.toLowerCase()) ||
1428
+ item.original.toLowerCase();
1429
+ matches.add(originalWord);
1430
+ if (!actualMatches.has(originalWord)) {
1431
+ actualMatches.set(originalWord, []);
1432
+ }
1433
+ actualMatches.get(originalWord)?.push(item.word);
1434
+ });
1435
+ }
1436
+ else {
1437
+ const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
1438
+ possibleProfanity.forEach((item) => {
1439
+ matches.add(item.original.toLowerCase());
1440
+ const originalWord = aliasMap.get(item.original.toLowerCase()) ||
1441
+ item.original.toLowerCase();
1442
+ if (!actualMatches.has(originalWord)) {
1443
+ actualMatches.set(originalWord, []);
1444
+ }
1445
+ actualMatches.get(originalWord)?.push(item.word);
1446
+ });
1447
+ }
1220
1448
  }
1449
+ findProfanity.lastActualMatches = actualMatches;
1221
1450
  return Array.from(matches);
1222
1451
  }
1223
1452
  /**
@@ -1307,14 +1536,20 @@ function calculateSeverity(matchDetails) {
1307
1536
  * @returns FilterResult dengan hasil filter
1308
1537
  */
1309
1538
  function filter(text, options = {}) {
1310
- const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, } = { ...DEFAULT_OPTIONS, ...options };
1539
+ const { replaceWith = '*', fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, detectSplit = false, detectSimilarity = false, useLevenshtein = false, maxLevenshteinDistance = 2, similarityThreshold = 0.8, } = { ...DEFAULT_OPTIONS, ...options };
1311
1540
  const matches = findProfanity(text, {
1312
1541
  ...options,
1313
1542
  detectLeetSpeak,
1314
1543
  whitelist,
1315
1544
  checkSubstring,
1316
1545
  indonesianVariation,
1546
+ detectSplit,
1547
+ detectSimilarity,
1548
+ useLevenshtein,
1549
+ maxLevenshteinDistance,
1550
+ similarityThreshold,
1317
1551
  });
1552
+ const actualMatches = findProfanity.lastActualMatches || new Map();
1318
1553
  const matchDetails = findProfanityWithMetadata(text, options);
1319
1554
  if (matches.length === 0) {
1320
1555
  return {
@@ -1329,36 +1564,120 @@ function filter(text, options = {}) {
1329
1564
  const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
1330
1565
  (m.aliases &&
1331
1566
  m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
1332
- const regex = createWordRegex(word, {
1333
- wholeWord: true,
1334
- caseSensitive: false,
1335
- leetSpeak: false,
1336
- detectSplit: false,
1337
- indonesianVariation: false,
1567
+ const variants = actualMatches.get(word.toLowerCase()) || [];
1568
+ variants.push(word);
1569
+ const uniqueVariants = [...new Set(variants)];
1570
+ uniqueVariants.forEach((variant) => {
1571
+ const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, 'gi');
1572
+ let match;
1573
+ while ((match = regex.exec(filteredText)) !== null) {
1574
+ const originalWord = match[0];
1575
+ if (whitelist.includes(originalWord.toLowerCase()))
1576
+ continue;
1577
+ let censoredWord;
1578
+ if (useRandomGrawlix) {
1579
+ censoredWord = makeRandomGrawlixString(originalWord.length);
1580
+ }
1581
+ else {
1582
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1583
+ }
1584
+ replacements.push({
1585
+ original: originalWord,
1586
+ censored: censoredWord,
1587
+ metadata,
1588
+ });
1589
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'), censoredWord);
1590
+ }
1338
1591
  });
1339
- let match;
1340
- const textToSearch = filteredText;
1341
- regex.lastIndex = 0;
1342
- while ((match = regex.exec(textToSearch)) !== null) {
1343
- const originalWord = match[0];
1344
- if (whitelist.includes(originalWord.toLowerCase()))
1345
- continue;
1346
- let censoredWord;
1347
- if (useRandomGrawlix) {
1348
- censoredWord = makeRandomGrawlixString(originalWord.length);
1592
+ if (detectSplit || detectLeetSpeak) {
1593
+ if (detectLeetSpeak) {
1594
+ const leetRegex = createWordRegex(word, {
1595
+ wholeWord: true,
1596
+ caseSensitive: false,
1597
+ leetSpeak: true,
1598
+ detectSplit: false,
1599
+ indonesianVariation: false,
1600
+ });
1601
+ let match;
1602
+ while ((match = leetRegex.exec(filteredText)) !== null) {
1603
+ const originalWord = match[0];
1604
+ if (whitelist.includes(originalWord.toLowerCase()))
1605
+ continue;
1606
+ let censoredWord;
1607
+ if (useRandomGrawlix) {
1608
+ censoredWord = makeRandomGrawlixString(originalWord.length);
1609
+ }
1610
+ else {
1611
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1612
+ }
1613
+ replacements.push({
1614
+ original: originalWord,
1615
+ censored: censoredWord,
1616
+ metadata,
1617
+ });
1618
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), 'g'), censoredWord);
1619
+ }
1349
1620
  }
1350
- else {
1351
- censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1621
+ if (detectSplit) {
1622
+ const splitRegex = createWordRegex(word, {
1623
+ wholeWord: false,
1624
+ caseSensitive: false,
1625
+ leetSpeak: false,
1626
+ detectSplit: true,
1627
+ indonesianVariation: false,
1628
+ });
1629
+ let match;
1630
+ while ((match = splitRegex.exec(filteredText)) !== null) {
1631
+ const originalWord = match[0];
1632
+ if (whitelist.includes(originalWord.toLowerCase()))
1633
+ continue;
1634
+ let censoredWord;
1635
+ if (useRandomGrawlix) {
1636
+ censoredWord = makeRandomGrawlixString(originalWord.length);
1637
+ }
1638
+ else {
1639
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1640
+ }
1641
+ replacements.push({
1642
+ original: originalWord,
1643
+ censored: censoredWord,
1644
+ metadata,
1645
+ });
1646
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), 'g'), censoredWord);
1647
+ }
1352
1648
  }
1353
- replacements.push({
1354
- original: originalWord,
1355
- censored: censoredWord,
1356
- metadata,
1357
- });
1358
- const replaceRegex = new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g");
1359
- filteredText = filteredText.replace(replaceRegex, censoredWord);
1360
1649
  }
1361
1650
  });
1651
+ if (detectSimilarity && useLevenshtein) {
1652
+ matches.forEach((word) => {
1653
+ const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
1654
+ (m.aliases &&
1655
+ m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
1656
+ const variants = actualMatches.get(word.toLowerCase()) || [];
1657
+ variants.forEach((variant) => {
1658
+ const exactVariantRegex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, 'gi');
1659
+ let match;
1660
+ while ((match = exactVariantRegex.exec(filteredText)) !== null) {
1661
+ const originalWord = match[0];
1662
+ if (whitelist.includes(originalWord.toLowerCase()))
1663
+ continue;
1664
+ let censoredWord;
1665
+ if (useRandomGrawlix) {
1666
+ censoredWord = makeRandomGrawlixString(originalWord.length);
1667
+ }
1668
+ else {
1669
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1670
+ }
1671
+ replacements.push({
1672
+ original: originalWord,
1673
+ censored: censoredWord,
1674
+ metadata,
1675
+ });
1676
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'), censoredWord);
1677
+ }
1678
+ });
1679
+ });
1680
+ }
1362
1681
  return {
1363
1682
  filtered: filteredText,
1364
1683
  censored: replacements.length,
@@ -1501,9 +1820,9 @@ function analyzeWithContext(text, contextWindowSize = 5, options = {}) {
1501
1820
  const regex = createContextRegex(word, contextWindowSize);
1502
1821
  let match;
1503
1822
  while ((match = regex.exec(text)) !== null) {
1504
- const beforeContext = match[1] || "";
1823
+ const beforeContext = match[1] || '';
1505
1824
  const wordMatch = match[2];
1506
- const afterContext = match[3] || "";
1825
+ const afterContext = match[3] || '';
1507
1826
  result.push({
1508
1827
  word: wordMatch,
1509
1828
  context: beforeContext + wordMatch + afterContext,
@@ -1633,10 +1952,25 @@ class IDProfanityFilter {
1633
1952
  /**
1634
1953
  * Mengaktifkan deteksi berdasarkan kesamaan
1635
1954
  * @param threshold Threshold kesamaan (0-1)
1955
+ * @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
1956
+ * @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
1957
+ */
1958
+ enableSimilarityDetection(threshold = 0.8, useLevenshtein = false, maxLevenshteinDistance = 2) {
1959
+ this.options.detectSimilarity = true;
1960
+ this.options.similarityThreshold = threshold;
1961
+ this.options.useLevenshtein = useLevenshtein;
1962
+ this.options.maxLevenshteinDistance = maxLevenshteinDistance;
1963
+ }
1964
+ /**
1965
+ * Mengaktifkan deteksi berbasis Levenshtein distance
1966
+ * @param threshold Threshold kesamaan (0-1)
1967
+ * @param maxDistance Jarak maksimal Levenshtein (default: 2)
1636
1968
  */
1637
- enableSimilarityDetection(threshold = 0.8) {
1969
+ enableLevenshteinDetection(threshold = 0.8, maxDistance = 2) {
1638
1970
  this.options.detectSimilarity = true;
1971
+ this.options.useLevenshtein = true;
1639
1972
  this.options.similarityThreshold = threshold;
1973
+ this.options.maxLevenshteinDistance = maxDistance;
1640
1974
  }
1641
1975
  }
1642
1976
  const idFilter = {
@@ -1682,8 +2016,10 @@ exports.escapeRegExp = escapeRegExp;
1682
2016
  exports.filter = filter;
1683
2017
  exports.findCategories = findCategories;
1684
2018
  exports.findMostSimilar = findMostSimilar;
2019
+ exports.findMostSimilarWithLevenshtein = findMostSimilarWithLevenshtein;
1685
2020
  exports.findPossibleProfanityBySimiliarity = findPossibleProfanityBySimiliarity;
1686
2021
  exports.findProfanity = findProfanity;
2022
+ exports.findProfanityByLevenshteinDistance = findProfanityByLevenshteinDistance;
1687
2023
  exports.findProfanityWithMetadata = findProfanityWithMetadata;
1688
2024
  exports.findRegions = findRegions;
1689
2025
  exports.getContextAroundIndex = getContextAroundIndex;