@sideid/id-profanity-filter 1.9.6 → 1.10.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.esm.js CHANGED
@@ -4,7 +4,7 @@ const general = [
4
4
  category: "profanity",
5
5
  region: "general",
6
6
  severity: 0.7,
7
- aliases: ["anjay", "anjir", "anying", "njing", "anj"],
7
+ aliases: ["anjay", "anjir", "anying", "njing", "anj", "anjg", "ajg"],
8
8
  description: "Mengacu pada hewan anjing, digunakan sebagai umpatan",
9
9
  context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
10
10
  },
@@ -13,7 +13,7 @@ const general = [
13
13
  category: "profanity",
14
14
  region: "general",
15
15
  severity: 0.6,
16
- aliases: ["bab1", "b4b1"],
16
+ aliases: ["bab1", "b4b1", "b4bi", "8481", "8ab1", "ba81"],
17
17
  description: "Mengacu pada hewan babi, digunakan sebagai umpatan",
18
18
  context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
19
19
  },
@@ -98,6 +98,105 @@ const general = [
98
98
  description: "Kata yang mengacu pada orang yang banyak bicara",
99
99
  context: "Hinaan untuk menyebut orang yang banyak bicara atau cerewet",
100
100
  },
101
+ {
102
+ word: "ngentot",
103
+ category: "sexual",
104
+ region: "general",
105
+ severity: 0.9,
106
+ aliases: ["ngentod", "ntot", "tod"],
107
+ description: "Istilah kasar untuk aktivitas seksual",
108
+ context: "Kata vulgar yang merujuk pada aktivitas seksual",
109
+ },
110
+ {
111
+ word: "sialan",
112
+ category: "insult",
113
+ region: "general",
114
+ severity: 0.5,
115
+ aliases: ["sialn", "sl"],
116
+ description: "Kata yang mengacu pada orang yang membawa sial",
117
+ context: "Hinaan untuk menyebut orang yang dianggap membawa sial",
118
+ },
119
+ {
120
+ word: "pler",
121
+ category: "sexual",
122
+ region: "general",
123
+ severity: 0.9,
124
+ aliases: ["peler", "plr", "biji"],
125
+ description: "Istilah kasar untuk alat kelamin laki-laki",
126
+ context: "Kata vulgar yang merujuk pada alat kelamin laki-laki",
127
+ },
128
+ {
129
+ word: "bokep",
130
+ category: "sexual",
131
+ region: "general",
132
+ severity: 0.7,
133
+ aliases: ["bkp", "bokap"],
134
+ description: "Istilah untuk video atau konten pornografi",
135
+ context: "Kata yang mengacu pada materi pornografi",
136
+ },
137
+ {
138
+ word: "coli",
139
+ category: "sexual",
140
+ region: "general",
141
+ severity: 0.8,
142
+ aliases: ["col", "coly"],
143
+ description: "Istilah untuk masturbasi laki-laki",
144
+ context: "Kata vulgar yang merujuk pada aktivitas seksual pribadi",
145
+ },
146
+ {
147
+ word: "desah",
148
+ category: "sexual",
149
+ region: "general",
150
+ severity: 0.6,
151
+ aliases: ["ds4h", "dsh"],
152
+ description: "Istilah untuk suara yang dibuat selama aktivitas seksual",
153
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
154
+ },
155
+ {
156
+ word: "seks",
157
+ category: "sexual",
158
+ region: "general",
159
+ severity: 0.5,
160
+ aliases: ["sex", "ML"],
161
+ description: "Istilah untuk aktivitas seksual",
162
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
163
+ },
164
+ {
165
+ word: "kondom",
166
+ category: "sexual",
167
+ region: "general",
168
+ severity: 0.5,
169
+ aliases: ["kndm", "kondom", "cd"],
170
+ description: "Alat kontrasepsi",
171
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
172
+ },
173
+ {
174
+ word: "ngewe",
175
+ category: "sexual",
176
+ region: "general",
177
+ severity: 0.9,
178
+ aliases: ["ngew", "we"],
179
+ description: "Istilah kasar untuk aktivitas seksual",
180
+ context: "Kata vulgar yang merujuk pada aktivitas seksual",
181
+ },
182
+ {
183
+ word: "puki",
184
+ category: "sexual",
185
+ region: "general",
186
+ severity: 0.9,
187
+ aliases: ["puk", "pukih"],
188
+ description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
189
+ context: "Kata vulgar yang merujuk pada anatomi seksual",
190
+ },
191
+ {
192
+ word: "xxx",
193
+ category: "sexual",
194
+ region: "general",
195
+ severity: 0.6,
196
+ aliases: ["xXx", "triplex"],
197
+ description: "Simbol yang sering digunakan untuk menandai konten pornografi",
198
+ context: "Digunakan untuk menandai konten seksual eksplisit",
199
+ },
101
200
  ];
102
201
  general.map((item) => item.word);
103
202
 
@@ -192,6 +291,15 @@ const jawa = [
192
291
  description: "Secara harfiah berarti kepalamu, digunakan sebagai umpatan ringan",
193
292
  context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
194
293
  },
294
+ {
295
+ word: "itil",
296
+ category: "sexual",
297
+ region: "jawa",
298
+ severity: 0.9,
299
+ aliases: ["itl", "itul"],
300
+ description: "Kata vulgar yang mengacu pada bagian dari alat kelamin perempuan",
301
+ context: "Kata vulgar yang merujuk pada anatomi seksual",
302
+ },
195
303
  ];
196
304
  jawa.map((item) => item.word);
197
305
 
@@ -712,7 +820,6 @@ function addLeetSpeakVariations(pattern) {
712
820
  t: ["t", "7", "+"],
713
821
  z: ["z", "2"],
714
822
  };
715
- // Ganti tiap karakter dengan variasinya dalam grup character class
716
823
  return pattern
717
824
  .split("")
718
825
  .map((char) => {
@@ -896,6 +1003,37 @@ function findMostSimilar(target, candidates, threshold = 0.7) {
896
1003
  }
897
1004
  return mostSimilar;
898
1005
  }
1006
+ /**
1007
+ * Mencari string yang paling mirip dari array menggunakan Levenshtein distance
1008
+ *
1009
+ * @param target String target
1010
+ * @param candidates Array string kandidat
1011
+ * @param threshold Minimum kesamaan yang diterima (0-1)
1012
+ * @param maxDistance Jarak Levenshtein maksimal yang diterima (default: 3)
1013
+ * @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
1014
+ */
1015
+ function findMostSimilarWithLevenshtein(target, candidates, threshold = 0.7, maxDistance = 3) {
1016
+ if (!candidates.length)
1017
+ return null;
1018
+ let maxSimilarity = 0;
1019
+ let minDistance = Infinity;
1020
+ let mostSimilar = null;
1021
+ for (const candidate of candidates) {
1022
+ if (Math.abs(target.length - candidate.length) > maxDistance)
1023
+ continue;
1024
+ const distance = levenshteinDistance(target, candidate);
1025
+ const similarity = stringSimilarity(target, candidate);
1026
+ if ((similarity > maxSimilarity && similarity >= threshold) ||
1027
+ (similarity >= threshold && distance < minDistance)) {
1028
+ maxSimilarity = similarity;
1029
+ minDistance = distance;
1030
+ mostSimilar = candidate;
1031
+ if (distance <= 1 || similarity > 0.95)
1032
+ break;
1033
+ }
1034
+ }
1035
+ return mostSimilar;
1036
+ }
899
1037
  /**
900
1038
  * Cek apakah string mungkin merupakan variasi dari kata kotor
901
1039
  * menggunakan kesamaan string
@@ -954,10 +1092,8 @@ function clusterSimilarWords(words, threshold = 0.8) {
954
1092
  */
955
1093
  function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.8) {
956
1094
  const result = [];
957
- // Pisahkan teks menjadi kata-kata
958
1095
  const words = text.toLowerCase().split(/\s+/);
959
1096
  for (const word of words) {
960
- // Lewati kata-kata yang terlalu pendek
961
1097
  if (word.length < 3)
962
1098
  continue;
963
1099
  for (const profanity of profanityWords) {
@@ -974,6 +1110,41 @@ function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.
974
1110
  }
975
1111
  return result;
976
1112
  }
1113
+ /**
1114
+ * Cari kata-kata kotor yang mungkin dari teks menggunakan Levenshtein distance
1115
+ *
1116
+ * @param text Teks yang akan diperiksa
1117
+ * @param profanityWords Daftar kata kotor
1118
+ * @param threshold Batas minimum kesamaan (default: 0.8)
1119
+ * @param maxDistance Jarak Levenshtein maksimal (default: 2)
1120
+ * @returns Array kata yang mungkin merupakan kata kotor
1121
+ */
1122
+ function findProfanityByLevenshteinDistance(text, profanityWords, threshold = 0.8, maxDistance = 2) {
1123
+ const result = [];
1124
+ const words = text.toLowerCase().split(/\s+/);
1125
+ for (const word of words) {
1126
+ if (word.length < 3)
1127
+ continue;
1128
+ for (const profanity of profanityWords) {
1129
+ if (Math.abs(word.length - profanity.length) > maxDistance)
1130
+ continue;
1131
+ const distance = levenshteinDistance(word, profanity);
1132
+ if (distance <= maxDistance) {
1133
+ const similarity = stringSimilarity(word, profanity);
1134
+ if (similarity >= threshold) {
1135
+ result.push({
1136
+ word,
1137
+ original: profanity,
1138
+ similarity,
1139
+ distance,
1140
+ });
1141
+ break;
1142
+ }
1143
+ }
1144
+ }
1145
+ }
1146
+ return result;
1147
+ }
977
1148
 
978
1149
  const DEFAULT_OPTIONS = {
979
1150
  replaceWith: "*",
@@ -982,6 +1153,8 @@ const DEFAULT_OPTIONS = {
982
1153
  checkSubstring: false,
983
1154
  whitelist: [],
984
1155
  severityThreshold: 0,
1156
+ useLevenshtein: false,
1157
+ maxLevenshteinDistance: 2,
985
1158
  };
986
1159
  const FILTER_PRESETS = {
987
1160
  strict: {
@@ -1127,31 +1300,44 @@ function makeRandomGrawlixString(length) {
1127
1300
  }
1128
1301
 
1129
1302
  function findProfanity(text, options = {}) {
1130
- const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, } = { ...DEFAULT_OPTIONS, ...options };
1303
+ const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, useLevenshtein = false, maxLevenshteinDistance = 2, } = { ...DEFAULT_OPTIONS, ...options };
1131
1304
  const normalizedText = normalizeText(text);
1132
- let wordsToCheck = wordList.length > 0 ? wordList : [];
1133
- if (wordsToCheck.length === 0) {
1134
- if (categories || regions || severityThreshold > 0) {
1135
- wordsToCheck = wordObjects
1136
- .filter((word) => {
1137
- const matchCategory = categories
1138
- ? categories.includes(word.category)
1139
- : true;
1140
- const matchRegion = regions ? regions.includes(word.region) : true;
1141
- const matchSeverity = word.severity >= severityThreshold;
1142
- return matchCategory && matchRegion && matchSeverity;
1143
- })
1144
- .map((word) => word.word);
1145
- }
1146
- else {
1147
- wordsToCheck = wordObjects.map((word) => word.word);
1148
- }
1305
+ let baseWordsToCheck = wordList.length > 0 ? wordList : [];
1306
+ if (baseWordsToCheck.length === 0) {
1307
+ const filteredWords = wordObjects.filter((word) => {
1308
+ const matchCategory = categories
1309
+ ? categories.includes(word.category)
1310
+ : true;
1311
+ const matchRegion = regions ? regions.includes(word.region) : true;
1312
+ const matchSeverity = word.severity >= severityThreshold;
1313
+ return matchCategory && matchRegion && matchSeverity;
1314
+ });
1315
+ baseWordsToCheck = filteredWords.map((word) => word.word);
1149
1316
  }
1150
- wordsToCheck = wordsToCheck.filter((word) => !whitelist.includes(word.toLocaleLowerCase()));
1317
+ const aliasMap = new Map();
1318
+ wordObjects.forEach((wordObj) => {
1319
+ if (wordObj.aliases && wordObj.aliases.length > 0) {
1320
+ const matchCategory = categories
1321
+ ? categories.includes(wordObj.category)
1322
+ : true;
1323
+ const matchRegion = regions ? regions.includes(wordObj.region) : true;
1324
+ const matchSeverity = wordObj.severity >= severityThreshold;
1325
+ if (matchCategory && matchRegion && matchSeverity) {
1326
+ wordObj.aliases.forEach((alias) => {
1327
+ aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
1328
+ });
1329
+ }
1330
+ }
1331
+ });
1332
+ const wordsToCheck = [
1333
+ ...baseWordsToCheck,
1334
+ ...Array.from(aliasMap.keys()),
1335
+ ].filter((word) => !whitelist.includes(word.toLowerCase()));
1151
1336
  if (wordsToCheck.length === 0) {
1152
1337
  return [];
1153
1338
  }
1154
1339
  const matches = new Set();
1340
+ const actualMatches = new Map();
1155
1341
  wordsToCheck.forEach((word) => {
1156
1342
  const regex = createWordRegex(word, {
1157
1343
  wholeWord: !checkSubstring,
@@ -1160,8 +1346,14 @@ function findProfanity(text, options = {}) {
1160
1346
  detectSplit: false,
1161
1347
  indonesianVariation: false,
1162
1348
  });
1163
- while ((regex.exec(normalizedText)) !== null) {
1164
- matches.add(word.toLowerCase());
1349
+ let match;
1350
+ while ((match = regex.exec(normalizedText)) !== null) {
1351
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1352
+ matches.add(originalWord);
1353
+ if (!actualMatches.has(originalWord)) {
1354
+ actualMatches.set(originalWord, []);
1355
+ }
1356
+ actualMatches.get(originalWord)?.push(match[0]);
1165
1357
  }
1166
1358
  });
1167
1359
  if (detectLeetSpeak) {
@@ -1173,8 +1365,14 @@ function findProfanity(text, options = {}) {
1173
1365
  detectSplit: false,
1174
1366
  indonesianVariation: false,
1175
1367
  });
1176
- while ((leetRegex.exec(text)) !== null) {
1177
- matches.add(word.toLowerCase());
1368
+ let match;
1369
+ while ((match = leetRegex.exec(text)) !== null) {
1370
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1371
+ matches.add(originalWord);
1372
+ if (!actualMatches.has(originalWord)) {
1373
+ actualMatches.set(originalWord, []);
1374
+ }
1375
+ actualMatches.get(originalWord)?.push(match[0]);
1178
1376
  }
1179
1377
  });
1180
1378
  }
@@ -1187,33 +1385,64 @@ function findProfanity(text, options = {}) {
1187
1385
  detectSplit: false,
1188
1386
  indonesianVariation: true,
1189
1387
  });
1190
- while ((variantRegex.exec(text)) !== null) {
1191
- matches.add(word.toLowerCase());
1388
+ let match;
1389
+ while ((match = variantRegex.exec(text)) !== null) {
1390
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1391
+ matches.add(originalWord);
1392
+ if (!actualMatches.has(originalWord)) {
1393
+ actualMatches.set(originalWord, []);
1394
+ }
1395
+ actualMatches.get(originalWord)?.push(match[0]);
1192
1396
  }
1193
1397
  });
1194
1398
  }
1195
1399
  if (detectSplit) {
1196
- if (detectSplitWords(text, wordsToCheck)) {
1197
- wordsToCheck.forEach((word) => {
1198
- const splitRegex = createWordRegex(word, {
1199
- wholeWord: false,
1200
- caseSensitive: false,
1201
- leetSpeak: false,
1202
- detectSplit: true,
1203
- indonesianVariation: false,
1204
- });
1205
- if (splitRegex.test(text)) {
1206
- matches.add(word.toLowerCase());
1207
- }
1400
+ wordsToCheck.forEach((word) => {
1401
+ const splitRegex = createWordRegex(word, {
1402
+ wholeWord: false,
1403
+ caseSensitive: false,
1404
+ leetSpeak: false,
1405
+ detectSplit: true,
1406
+ indonesianVariation: false,
1208
1407
  });
1209
- }
1408
+ let match;
1409
+ while ((match = splitRegex.exec(text)) !== null) {
1410
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1411
+ matches.add(originalWord);
1412
+ if (!actualMatches.has(originalWord)) {
1413
+ actualMatches.set(originalWord, []);
1414
+ }
1415
+ actualMatches.get(originalWord)?.push(match[0]);
1416
+ }
1417
+ });
1210
1418
  }
1211
1419
  if (detectSimilarity) {
1212
- const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
1213
- possibleProfanity.forEach((item) => {
1214
- matches.add(item.original.toLowerCase());
1215
- });
1420
+ if (useLevenshtein) {
1421
+ const possibleProfanity = findProfanityByLevenshteinDistance(text, wordsToCheck, similarityThreshold, maxLevenshteinDistance);
1422
+ possibleProfanity.forEach((item) => {
1423
+ const originalWord = aliasMap.get(item.original.toLowerCase()) ||
1424
+ item.original.toLowerCase();
1425
+ matches.add(originalWord);
1426
+ if (!actualMatches.has(originalWord)) {
1427
+ actualMatches.set(originalWord, []);
1428
+ }
1429
+ actualMatches.get(originalWord)?.push(item.word);
1430
+ });
1431
+ }
1432
+ else {
1433
+ const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
1434
+ possibleProfanity.forEach((item) => {
1435
+ matches.add(item.original.toLowerCase());
1436
+ const originalWord = aliasMap.get(item.original.toLowerCase()) ||
1437
+ item.original.toLowerCase();
1438
+ if (!actualMatches.has(originalWord)) {
1439
+ actualMatches.set(originalWord, []);
1440
+ }
1441
+ actualMatches.get(originalWord)?.push(item.word);
1442
+ });
1443
+ }
1216
1444
  }
1445
+ findProfanity.lastActualMatches = actualMatches;
1217
1446
  return Array.from(matches);
1218
1447
  }
1219
1448
  /**
@@ -1303,14 +1532,20 @@ function calculateSeverity(matchDetails) {
1303
1532
  * @returns FilterResult dengan hasil filter
1304
1533
  */
1305
1534
  function filter(text, options = {}) {
1306
- const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, } = { ...DEFAULT_OPTIONS, ...options };
1535
+ const { replaceWith = '*', fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, detectSplit = false, detectSimilarity = false, useLevenshtein = false, maxLevenshteinDistance = 2, similarityThreshold = 0.8, } = { ...DEFAULT_OPTIONS, ...options };
1307
1536
  const matches = findProfanity(text, {
1308
1537
  ...options,
1309
1538
  detectLeetSpeak,
1310
1539
  whitelist,
1311
1540
  checkSubstring,
1312
1541
  indonesianVariation,
1542
+ detectSplit,
1543
+ detectSimilarity,
1544
+ useLevenshtein,
1545
+ maxLevenshteinDistance,
1546
+ similarityThreshold,
1313
1547
  });
1548
+ const actualMatches = findProfanity.lastActualMatches || new Map();
1314
1549
  const matchDetails = findProfanityWithMetadata(text, options);
1315
1550
  if (matches.length === 0) {
1316
1551
  return {
@@ -1325,36 +1560,120 @@ function filter(text, options = {}) {
1325
1560
  const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
1326
1561
  (m.aliases &&
1327
1562
  m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
1328
- const regex = createWordRegex(word, {
1329
- wholeWord: true,
1330
- caseSensitive: false,
1331
- leetSpeak: false,
1332
- detectSplit: false,
1333
- indonesianVariation: false,
1563
+ const variants = actualMatches.get(word.toLowerCase()) || [];
1564
+ variants.push(word);
1565
+ const uniqueVariants = [...new Set(variants)];
1566
+ uniqueVariants.forEach((variant) => {
1567
+ const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, 'gi');
1568
+ let match;
1569
+ while ((match = regex.exec(filteredText)) !== null) {
1570
+ const originalWord = match[0];
1571
+ if (whitelist.includes(originalWord.toLowerCase()))
1572
+ continue;
1573
+ let censoredWord;
1574
+ if (useRandomGrawlix) {
1575
+ censoredWord = makeRandomGrawlixString(originalWord.length);
1576
+ }
1577
+ else {
1578
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1579
+ }
1580
+ replacements.push({
1581
+ original: originalWord,
1582
+ censored: censoredWord,
1583
+ metadata,
1584
+ });
1585
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'), censoredWord);
1586
+ }
1334
1587
  });
1335
- let match;
1336
- const textToSearch = filteredText;
1337
- regex.lastIndex = 0;
1338
- while ((match = regex.exec(textToSearch)) !== null) {
1339
- const originalWord = match[0];
1340
- if (whitelist.includes(originalWord.toLowerCase()))
1341
- continue;
1342
- let censoredWord;
1343
- if (useRandomGrawlix) {
1344
- censoredWord = makeRandomGrawlixString(originalWord.length);
1588
+ if (detectSplit || detectLeetSpeak) {
1589
+ if (detectLeetSpeak) {
1590
+ const leetRegex = createWordRegex(word, {
1591
+ wholeWord: true,
1592
+ caseSensitive: false,
1593
+ leetSpeak: true,
1594
+ detectSplit: false,
1595
+ indonesianVariation: false,
1596
+ });
1597
+ let match;
1598
+ while ((match = leetRegex.exec(filteredText)) !== null) {
1599
+ const originalWord = match[0];
1600
+ if (whitelist.includes(originalWord.toLowerCase()))
1601
+ continue;
1602
+ let censoredWord;
1603
+ if (useRandomGrawlix) {
1604
+ censoredWord = makeRandomGrawlixString(originalWord.length);
1605
+ }
1606
+ else {
1607
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1608
+ }
1609
+ replacements.push({
1610
+ original: originalWord,
1611
+ censored: censoredWord,
1612
+ metadata,
1613
+ });
1614
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), 'g'), censoredWord);
1615
+ }
1345
1616
  }
1346
- else {
1347
- censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1617
+ if (detectSplit) {
1618
+ const splitRegex = createWordRegex(word, {
1619
+ wholeWord: false,
1620
+ caseSensitive: false,
1621
+ leetSpeak: false,
1622
+ detectSplit: true,
1623
+ indonesianVariation: false,
1624
+ });
1625
+ let match;
1626
+ while ((match = splitRegex.exec(filteredText)) !== null) {
1627
+ const originalWord = match[0];
1628
+ if (whitelist.includes(originalWord.toLowerCase()))
1629
+ continue;
1630
+ let censoredWord;
1631
+ if (useRandomGrawlix) {
1632
+ censoredWord = makeRandomGrawlixString(originalWord.length);
1633
+ }
1634
+ else {
1635
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1636
+ }
1637
+ replacements.push({
1638
+ original: originalWord,
1639
+ censored: censoredWord,
1640
+ metadata,
1641
+ });
1642
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), 'g'), censoredWord);
1643
+ }
1348
1644
  }
1349
- replacements.push({
1350
- original: originalWord,
1351
- censored: censoredWord,
1352
- metadata,
1353
- });
1354
- const replaceRegex = new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g");
1355
- filteredText = filteredText.replace(replaceRegex, censoredWord);
1356
1645
  }
1357
1646
  });
1647
+ if (detectSimilarity && useLevenshtein) {
1648
+ matches.forEach((word) => {
1649
+ const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
1650
+ (m.aliases &&
1651
+ m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
1652
+ const variants = actualMatches.get(word.toLowerCase()) || [];
1653
+ variants.forEach((variant) => {
1654
+ const exactVariantRegex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, 'gi');
1655
+ let match;
1656
+ while ((match = exactVariantRegex.exec(filteredText)) !== null) {
1657
+ const originalWord = match[0];
1658
+ if (whitelist.includes(originalWord.toLowerCase()))
1659
+ continue;
1660
+ let censoredWord;
1661
+ if (useRandomGrawlix) {
1662
+ censoredWord = makeRandomGrawlixString(originalWord.length);
1663
+ }
1664
+ else {
1665
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
1666
+ }
1667
+ replacements.push({
1668
+ original: originalWord,
1669
+ censored: censoredWord,
1670
+ metadata,
1671
+ });
1672
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'), censoredWord);
1673
+ }
1674
+ });
1675
+ });
1676
+ }
1358
1677
  return {
1359
1678
  filtered: filteredText,
1360
1679
  censored: replacements.length,
@@ -1497,9 +1816,9 @@ function analyzeWithContext(text, contextWindowSize = 5, options = {}) {
1497
1816
  const regex = createContextRegex(word, contextWindowSize);
1498
1817
  let match;
1499
1818
  while ((match = regex.exec(text)) !== null) {
1500
- const beforeContext = match[1] || "";
1819
+ const beforeContext = match[1] || '';
1501
1820
  const wordMatch = match[2];
1502
- const afterContext = match[3] || "";
1821
+ const afterContext = match[3] || '';
1503
1822
  result.push({
1504
1823
  word: wordMatch,
1505
1824
  context: beforeContext + wordMatch + afterContext,
@@ -1629,10 +1948,25 @@ class IDProfanityFilter {
1629
1948
  /**
1630
1949
  * Mengaktifkan deteksi berdasarkan kesamaan
1631
1950
  * @param threshold Threshold kesamaan (0-1)
1951
+ * @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
1952
+ * @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
1953
+ */
1954
+ enableSimilarityDetection(threshold = 0.8, useLevenshtein = false, maxLevenshteinDistance = 2) {
1955
+ this.options.detectSimilarity = true;
1956
+ this.options.similarityThreshold = threshold;
1957
+ this.options.useLevenshtein = useLevenshtein;
1958
+ this.options.maxLevenshteinDistance = maxLevenshteinDistance;
1959
+ }
1960
+ /**
1961
+ * Mengaktifkan deteksi berbasis Levenshtein distance
1962
+ * @param threshold Threshold kesamaan (0-1)
1963
+ * @param maxDistance Jarak maksimal Levenshtein (default: 2)
1632
1964
  */
1633
- enableSimilarityDetection(threshold = 0.8) {
1965
+ enableLevenshteinDetection(threshold = 0.8, maxDistance = 2) {
1634
1966
  this.options.detectSimilarity = true;
1967
+ this.options.useLevenshtein = true;
1635
1968
  this.options.similarityThreshold = threshold;
1969
+ this.options.maxLevenshteinDistance = maxDistance;
1636
1970
  }
1637
1971
  }
1638
1972
  const idFilter = {
@@ -1648,5 +1982,5 @@ const idFilter = {
1648
1982
  },
1649
1983
  };
1650
1984
 
1651
- export { CATEGORY_PRESETS, DEFAULT_OPTIONS, FILTER_PRESETS, IDProfanityFilter, REGION_PRESETS, REPLACEMENT_CHARS, addIndonesianVariations, addLeetSpeakVariations, addSplitVariations, analyze, analyzeBySentence, analyzeWithContext, batchAnalyze, calculateSeverity, censorWord, clusterSimilarWords, containsAnyWord, containsEuphemism, createContextRegex, createEvasionRegex, createIndonesianVariationRegex, createOptions, createWordFormRegex, createWordRegex, IDProfanityFilter as default, detectSplitWords, escapeRegExp, filter, findCategories, findMostSimilar, findPossibleProfanityBySimiliarity, findProfanity, findProfanityWithMetadata, findRegions, getContextAroundIndex, getPresetOptions, getRandomGrawlix, getReplacementChar, idFilter, isPossibleProfanityVariation, isProfane, levenshteinDistance, makeRandomGrawlixString, maskText, normalizeText, splitIntoSentences, stringSimilarity, toLeetSpeak };
1985
+ export { CATEGORY_PRESETS, DEFAULT_OPTIONS, FILTER_PRESETS, IDProfanityFilter, REGION_PRESETS, REPLACEMENT_CHARS, addIndonesianVariations, addLeetSpeakVariations, addSplitVariations, analyze, analyzeBySentence, analyzeWithContext, batchAnalyze, calculateSeverity, censorWord, clusterSimilarWords, containsAnyWord, containsEuphemism, createContextRegex, createEvasionRegex, createIndonesianVariationRegex, createOptions, createWordFormRegex, createWordRegex, IDProfanityFilter as default, detectSplitWords, escapeRegExp, filter, findCategories, findMostSimilar, findMostSimilarWithLevenshtein, findPossibleProfanityBySimiliarity, findProfanity, findProfanityByLevenshteinDistance, findProfanityWithMetadata, findRegions, getContextAroundIndex, getPresetOptions, getRandomGrawlix, getReplacementChar, idFilter, isPossibleProfanityVariation, isProfane, levenshteinDistance, makeRandomGrawlixString, maskText, normalizeText, splitIntoSentences, stringSimilarity, toLeetSpeak };
1652
1986
  //# sourceMappingURL=index.esm.js.map