@sideid/id-profanity-filter 1.9.5 → 1.10.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/.eslintrc.js +44 -16
  2. package/.github/workflows/release.yml +62 -0
  3. package/CONTRIBUTING.md +150 -150
  4. package/LICENSE +21 -21
  5. package/README.md +548 -285
  6. package/dist/config/options.d.ts +24 -0
  7. package/dist/constants/categories/blasphemy.d.ts +4 -0
  8. package/dist/constants/categories/disgusting.d.ts +4 -0
  9. package/dist/constants/categories/drugs.d.ts +4 -0
  10. package/dist/constants/categories/profanity.d.ts +4 -0
  11. package/dist/constants/categories/slur.d.ts +4 -0
  12. package/dist/index.d.ts +2 -0
  13. package/dist/index.esm.js +923 -94
  14. package/dist/index.esm.js.map +1 -1
  15. package/dist/index.js +924 -93
  16. package/dist/index.js.map +1 -1
  17. package/dist/types/index.d.ts +2 -0
  18. package/dist/utils/ahoCorasick.d.ts +36 -0
  19. package/dist/utils/similarityUtils.d.ts +35 -0
  20. package/eslint.config.mjs +40 -0
  21. package/examples/advanced.ts +120 -0
  22. package/examples/basic.ts +71 -52
  23. package/examples/custom-list.ts +140 -0
  24. package/jest.config.mjs +10 -10
  25. package/package.json +3 -2
  26. package/prettierrc +6 -6
  27. package/rollup.config.mjs +35 -35
  28. package/src/config/options.ts +2 -0
  29. package/src/constants/categories/blasphemy.ts +25 -0
  30. package/src/constants/categories/disgusting.ts +82 -0
  31. package/src/constants/categories/drugs.ts +72 -0
  32. package/src/constants/categories/profanity.ts +139 -0
  33. package/src/constants/categories/slur.ts +102 -0
  34. package/src/constants/regions/general.ts +111 -2
  35. package/src/constants/regions/jawa.ts +257 -3
  36. package/src/constants/wordList.ts +15 -8
  37. package/src/core/analyzer.ts +28 -13
  38. package/src/core/filter.ts +178 -37
  39. package/src/core/matcher.ts +146 -69
  40. package/src/index.ts +21 -2
  41. package/src/types/index.ts +4 -2
  42. package/src/utils/ahoCorasick.ts +179 -0
  43. package/src/utils/regexUtils.ts +0 -1
  44. package/src/utils/similarityUtils.ts +239 -7
  45. package/tsconfig.json +115 -115
  46. package/.github/workflows/ci.yml +0 -0
  47. package/src/constants/categories/index.ts +0 -31
  48. package/src/constants/regions/index.ts +0 -62
package/dist/index.js CHANGED
@@ -8,7 +8,7 @@ const general = [
8
8
  category: "profanity",
9
9
  region: "general",
10
10
  severity: 0.7,
11
- aliases: ["anjay", "anjir", "anying", "njing", "anj"],
11
+ aliases: ["anjay", "anjir", "anying", "njing", "anj", "anjg", "ajg"],
12
12
  description: "Mengacu pada hewan anjing, digunakan sebagai umpatan",
13
13
  context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
14
14
  },
@@ -17,7 +17,7 @@ const general = [
17
17
  category: "profanity",
18
18
  region: "general",
19
19
  severity: 0.6,
20
- aliases: ["bab1", "b4b1"],
20
+ aliases: ["bab1", "b4b1", "b4bi", "8481", "8ab1", "ba81"],
21
21
  description: "Mengacu pada hewan babi, digunakan sebagai umpatan",
22
22
  context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
23
23
  },
@@ -102,6 +102,114 @@ const general = [
102
102
  description: "Kata yang mengacu pada orang yang banyak bicara",
103
103
  context: "Hinaan untuk menyebut orang yang banyak bicara atau cerewet",
104
104
  },
105
+ {
106
+ word: "ngentot",
107
+ category: "sexual",
108
+ region: "general",
109
+ severity: 0.9,
110
+ aliases: ["ngentod", "ntot", "tod"],
111
+ description: "Istilah kasar untuk aktivitas seksual",
112
+ context: "Kata vulgar yang merujuk pada aktivitas seksual",
113
+ },
114
+ {
115
+ word: "sialan",
116
+ category: "insult",
117
+ region: "general",
118
+ severity: 0.5,
119
+ aliases: ["sialn", "sl"],
120
+ description: "Kata yang mengacu pada orang yang membawa sial",
121
+ context: "Hinaan untuk menyebut orang yang dianggap membawa sial",
122
+ },
123
+ {
124
+ word: "pler",
125
+ category: "sexual",
126
+ region: "general",
127
+ severity: 0.9,
128
+ aliases: ["peler", "plr", "biji"],
129
+ description: "Istilah kasar untuk alat kelamin laki-laki",
130
+ context: "Kata vulgar yang merujuk pada alat kelamin laki-laki",
131
+ },
132
+ {
133
+ word: "bokep",
134
+ category: "sexual",
135
+ region: "general",
136
+ severity: 0.7,
137
+ aliases: ["bkp", "bokap"],
138
+ description: "Istilah untuk video atau konten pornografi",
139
+ context: "Kata yang mengacu pada materi pornografi",
140
+ },
141
+ {
142
+ word: "coli",
143
+ category: "sexual",
144
+ region: "general",
145
+ severity: 0.8,
146
+ aliases: ["col", "coly"],
147
+ description: "Istilah untuk masturbasi laki-laki",
148
+ context: "Kata vulgar yang merujuk pada aktivitas seksual pribadi",
149
+ },
150
+ {
151
+ word: "desah",
152
+ category: "sexual",
153
+ region: "general",
154
+ severity: 0.6,
155
+ aliases: ["ds4h", "dsh"],
156
+ description: "Istilah untuk suara yang dibuat selama aktivitas seksual",
157
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
158
+ },
159
+ {
160
+ word: "seks",
161
+ category: "sexual",
162
+ region: "general",
163
+ severity: 0.5,
164
+ aliases: ["sex", "ML"],
165
+ description: "Istilah untuk aktivitas seksual",
166
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
167
+ },
168
+ {
169
+ word: "kondom",
170
+ category: "sexual",
171
+ region: "general",
172
+ severity: 0.5,
173
+ aliases: ["kndm", "kondom", "cd"],
174
+ description: "Alat kontrasepsi",
175
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
176
+ },
177
+ {
178
+ word: "ngewe",
179
+ category: "sexual",
180
+ region: "general",
181
+ severity: 0.9,
182
+ aliases: ["ngew", "we"],
183
+ description: "Istilah kasar untuk aktivitas seksual",
184
+ context: "Kata vulgar yang merujuk pada aktivitas seksual",
185
+ },
186
+ {
187
+ word: "puki",
188
+ category: "sexual",
189
+ region: "general",
190
+ severity: 0.9,
191
+ aliases: ["puk", "pukih"],
192
+ description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
193
+ context: "Kata vulgar yang merujuk pada anatomi seksual",
194
+ },
195
+ {
196
+ word: "xxx",
197
+ category: "sexual",
198
+ region: "general",
199
+ severity: 0.6,
200
+ aliases: ["xXx", "triplex"],
201
+ description: "Simbol yang sering digunakan untuk menandai konten pornografi",
202
+ context: "Digunakan untuk menandai konten seksual eksplisit",
203
+ },
204
+ {
205
+ word: "xnxx",
206
+ category: "sexual",
207
+ region: "general",
208
+ severity: 0.6,
209
+ aliases: ["xnxx", "xnx"],
210
+ description: "simbol yang sering digunakan untuk menandai konten pornografi",
211
+ context: "Digunakan untuk menandai konten seksual eksplisit",
212
+ }
105
213
  ];
106
214
  general.map((item) => item.word);
107
215
 
@@ -119,7 +227,7 @@ const jawa = [
119
227
  word: "jancok",
120
228
  category: "sexual",
121
229
  region: "jawa",
122
- severity: 0.8,
230
+ severity: 0.9,
123
231
  aliases: ["jancuk", "jncok", "jancuk", "jncuk", "dancok", "dancuk"],
124
232
  description: "Kata umpatan kasar dalam Bahasa Jawa",
125
233
  context: "Umpatan kasar yang umum digunakan di Jawa Timur",
@@ -155,7 +263,7 @@ const jawa = [
155
263
  word: "mbokne ancok",
156
264
  category: "insult",
157
265
  region: "jawa",
158
- severity: 0.8,
266
+ severity: 0.9,
159
267
  aliases: ["mbokne", "mbokneancok"],
160
268
  description: "Umpatan yang menyinggung ibu seseorang",
161
269
  context: "Umpatan kasar yang menyinggung orangtua orang lain",
@@ -164,7 +272,7 @@ const jawa = [
164
272
  word: "pekok",
165
273
  category: "insult",
166
274
  region: "jawa",
167
- severity: 0.6,
275
+ severity: 0.7,
168
276
  aliases: ["pekak", "pekilk"],
169
277
  description: "Kata hinaan yang menunjukkan kebodohan",
170
278
  context: "Hinaan untuk menyebut orang yang dianggap sangat bodoh",
@@ -196,6 +304,257 @@ const jawa = [
196
304
  description: "Secara harfiah berarti kepalamu, digunakan sebagai umpatan ringan",
197
305
  context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
198
306
  },
307
+ {
308
+ word: "kontol",
309
+ category: "sexual",
310
+ region: "jawa",
311
+ severity: 0.8,
312
+ aliases: ["kntl", "kontl"],
313
+ description: "Mengacu ke alat kelamin laki-laki",
314
+ context: "Kata vulgar yang sering digunakan sebagai umpatan kasar",
315
+ },
316
+ {
317
+ word: "tempek",
318
+ category: "sexual",
319
+ region: "jawa",
320
+ severity: 0.8,
321
+ aliases: ["mpek", "torok", "tempk"],
322
+ description: "Mengacu pada alat kelamin perempuan",
323
+ context: "Kata vulgar yang digunakan sebagai umpatan atau hinaan",
324
+ },
325
+ {
326
+ word: "silit",
327
+ category: "insult",
328
+ region: "jawa",
329
+ severity: 0.6,
330
+ aliases: ["selet", "tilis"],
331
+ description: "Mengacu pada bagian dubur atau anus",
332
+ context: "Kata kasar yang digunakan sebagai hinaan",
333
+ },
334
+ {
335
+ word: "mbahmu",
336
+ category: "insult",
337
+ region: "jawa",
338
+ severity: 0.6,
339
+ aliases: ["mbahmu kiper", "mbah mu"],
340
+ description: "Hinaan yang menyinggung nenek/kakek seseorang",
341
+ context: "Umpatan ringan untuk menanggapi sesuatu yang tidak masuk akal",
342
+ },
343
+ {
344
+ word: "makmu",
345
+ category: "insult",
346
+ region: "jawa",
347
+ severity: 0.7,
348
+ aliases: ["mak mu", "mamamu"],
349
+ description: "Hinaan yang menyinggung ibu seseorang",
350
+ context: "Umpatan yang dianggap kasar karena menyinggung orang tua",
351
+ },
352
+ {
353
+ word: "bajingan",
354
+ category: "insult",
355
+ region: "jawa",
356
+ severity: 0.7,
357
+ aliases: [
358
+ "bajilak",
359
+ "bajhingan",
360
+ "bajingak",
361
+ "bajingseng",
362
+ "bajindul",
363
+ "bajigur",
364
+ "jingan",
365
+ ],
366
+ description: "Sebutan untuk orang yang dianggap jahat atau tidak bermoral",
367
+ context: "Umpatan untuk mengekspresikan kemarahan atau kekesalan",
368
+ },
369
+ {
370
+ word: "cocote",
371
+ category: "insult",
372
+ region: "jawa",
373
+ severity: 0.6,
374
+ aliases: ["cocot", "bacot", "nyocot"],
375
+ description: "Mengacu pada mulut dengan konotasi negatif",
376
+ context: "Umpatan untuk menyuruh seseorang berhenti berbicara",
377
+ },
378
+ {
379
+ word: "ngentot",
380
+ category: "sexual",
381
+ region: "jawa",
382
+ severity: 0.9,
383
+ aliases: ["kentu", "kentot", "iclik", "ngtt", "iclk"],
384
+ description: "Mengacu pada aktivitas seksual",
385
+ context: "Kata vulgar yang digunakan sebagai umpatan kasar",
386
+ },
387
+ {
388
+ word: "edan",
389
+ category: "insult",
390
+ region: "jawa",
391
+ severity: 0.5,
392
+ aliases: ["gendeng", "gila", "gendheng", "sarap"],
393
+ description: "Secara harfiah berarti gila atau tidak waras",
394
+ context: "Umpatan untuk menyebut seseorang yang dianggap tidak masuk akal",
395
+ },
396
+ {
397
+ word: "dapuranmu",
398
+ category: "insult",
399
+ region: "jawa",
400
+ severity: 0.6,
401
+ aliases: ["raimu", "rai mu"],
402
+ description: "Secara harfiah mengacu pada wajah atau rupa seseorang",
403
+ context: "Umpatan untuk menghina penampilan atau wajah seseorang",
404
+ },
405
+ {
406
+ word: "damput",
407
+ category: "insult",
408
+ region: "jawa",
409
+ severity: 0.7,
410
+ aliases: ["diamput"],
411
+ description: "Variasi bentuk umpatan dengan makna serupa dengan jancok",
412
+ context: "Umpatan kasar untuk mengekspresikan kemarahan",
413
+ },
414
+ {
415
+ word: "mbathang",
416
+ category: "insult",
417
+ region: "jawa",
418
+ severity: 0.7,
419
+ aliases: ["mbatang"],
420
+ description: "Secara harfiah berarti bangkai",
421
+ context: "Umpatan kasar untuk menghina seseorang",
422
+ },
423
+ {
424
+ word: "ndlogok",
425
+ category: "insult",
426
+ region: "jawa",
427
+ severity: 0.6,
428
+ aliases: ["ndelodok", "ndlodok"],
429
+ description: "Mengacu pada tindakan yang dianggap bodoh atau tidak masuk akal",
430
+ context: "Hinaan untuk mengkritik tindakan seseorang",
431
+ },
432
+ {
433
+ word: "nggateli",
434
+ category: "insult",
435
+ region: "jawa",
436
+ severity: 0.5,
437
+ aliases: ["gateli", "gathel"],
438
+ description: "Secara harfiah berarti gatal atau menyebalkan",
439
+ context: "Ungkapan untuk menunjukkan kekesalan terhadap perilaku seseorang",
440
+ },
441
+ {
442
+ word: "perek",
443
+ category: "sexual",
444
+ region: "jawa",
445
+ severity: 0.8,
446
+ aliases: ["lonthe", "pelacur"],
447
+ description: "Istilah merendahkan untuk pekerja seks komersial",
448
+ context: "Kata kasar untuk menghina wanita",
449
+ },
450
+ {
451
+ word: "picek",
452
+ category: "insult",
453
+ region: "jawa",
454
+ severity: 0.6,
455
+ aliases: ["pcek", "buta"],
456
+ description: "Secara harfiah berarti buta atau tidak bisa melihat",
457
+ context: "Hinaan untuk orang yang dianggap tidak bisa melihat kenyataan",
458
+ },
459
+ {
460
+ word: "untumu",
461
+ category: "insult",
462
+ region: "jawa",
463
+ severity: 0.5,
464
+ aliases: ["gigimu", "untu mu"],
465
+ description: "Secara harfiah berarti gigimu",
466
+ context: "Umpatan ringan untuk menanggapi sesuatu yang tidak disetujui",
467
+ },
468
+ {
469
+ word: "goblog",
470
+ category: "insult",
471
+ region: "jawa",
472
+ severity: 0.7,
473
+ aliases: ["ghoblog", "goblok", "gobhlok", "pekok"],
474
+ description: "Kata hinaan yang menunjukkan kebodohan ekstrem",
475
+ context: "Hinaan untuk menyebut seseorang yang dianggap sangat bodoh",
476
+ },
477
+ {
478
+ word: "tolol",
479
+ category: "insult",
480
+ region: "jawa",
481
+ severity: 0.7,
482
+ aliases: ["tholol", "tlol"],
483
+ description: "Kata hinaan yang menunjukkan kebodohan",
484
+ context: "Hinaan untuk menyebut seseorang yang dianggap bodoh",
485
+ },
486
+ {
487
+ word: "budheg",
488
+ category: "insult",
489
+ region: "jawa",
490
+ severity: 0.6,
491
+ aliases: ["budeg", "bdeg"],
492
+ description: "Secara harfiah berarti tuli atau tidak bisa mendengar",
493
+ context: "Hinaan untuk orang yang dianggap tidak mau mendengarkan",
494
+ },
495
+ {
496
+ word: "jiangkrik",
497
+ category: "insult",
498
+ region: "jawa",
499
+ severity: 0.4,
500
+ aliases: ["jiangkrek", "jangkrik"],
501
+ description: "Secara harfiah berarti jangkrik, digunakan sebagai eufemisme",
502
+ context: "Umpatan ringan sebagai pengganti kata kasar yang lebih vulgar",
503
+ },
504
+ {
505
+ word: "diamput",
506
+ category: "insult",
507
+ region: "jawa",
508
+ severity: 0.8,
509
+ aliases: ["damput", "djamput"],
510
+ description: "Bentuk umpatan kasar dengan makna serupa jancok",
511
+ context: "Kata kasar untuk mengekspresikan kemarahan",
512
+ },
513
+ {
514
+ word: "celeng",
515
+ category: "insult",
516
+ region: "jawa",
517
+ severity: 0.6,
518
+ aliases: ["cleng", "babi hutan"],
519
+ description: "Secara harfiah berarti babi hutan",
520
+ context: "Hinaan untuk orang yang dianggap jorok atau rakus",
521
+ },
522
+ {
523
+ word: "kampret",
524
+ category: "insult",
525
+ region: "jawa",
526
+ severity: 0.5,
527
+ aliases: ["kmpret", "kmprt"],
528
+ description: "Secara harfiah berarti kelelawar kecil",
529
+ context: "Umpatan ringan untuk mengekspresikan kekesalan",
530
+ },
531
+ {
532
+ word: "ndeso",
533
+ category: "insult",
534
+ region: "jawa",
535
+ severity: 0.4,
536
+ aliases: ["ndesa", "deso"],
537
+ description: "Secara harfiah berarti dari desa atau kampungan",
538
+ context: "Hinaan untuk orang yang dianggap kurang modern atau berpendidikan",
539
+ },
540
+ {
541
+ word: "kere",
542
+ category: "insult",
543
+ region: "jawa",
544
+ severity: 0.5,
545
+ aliases: ["miskin", "mlarat"],
546
+ description: "Secara harfiah berarti miskin atau tidak punya uang",
547
+ context: "Hinaan untuk status ekonomi seseorang yang dianggap rendah",
548
+ },
549
+ {
550
+ word: "itil",
551
+ category: "sexual",
552
+ region: "jawa",
553
+ severity: 0.9,
554
+ aliases: ["itl", "itul"],
555
+ description: "Kata vulgar yang mengacu pada bagian dari alat kelamin perempuan",
556
+ context: "Kata vulgar yang merujuk pada anatomi seksual",
557
+ },
199
558
  ];
200
559
  jawa.map((item) => item.word);
201
560
 
@@ -716,7 +1075,6 @@ function addLeetSpeakVariations(pattern) {
716
1075
  t: ["t", "7", "+"],
717
1076
  z: ["z", "2"],
718
1077
  };
719
- // Ganti tiap karakter dengan variasinya dalam grup character class
720
1078
  return pattern
721
1079
  .split("")
722
1080
  .map((char) => {
@@ -900,6 +1258,37 @@ function findMostSimilar(target, candidates, threshold = 0.7) {
900
1258
  }
901
1259
  return mostSimilar;
902
1260
  }
1261
+ /**
1262
+ * Mencari string yang paling mirip dari array menggunakan Levenshtein distance
1263
+ *
1264
+ * @param target String target
1265
+ * @param candidates Array string kandidat
1266
+ * @param threshold Minimum kesamaan yang diterima (0-1)
1267
+ * @param maxDistance Jarak Levenshtein maksimal yang diterima (default: 3)
1268
+ * @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
1269
+ */
1270
+ function findMostSimilarWithLevenshtein(target, candidates, threshold = 0.7, maxDistance = 3) {
1271
+ if (!candidates.length)
1272
+ return null;
1273
+ let maxSimilarity = 0;
1274
+ let minDistance = Infinity;
1275
+ let mostSimilar = null;
1276
+ for (const candidate of candidates) {
1277
+ if (Math.abs(target.length - candidate.length) > maxDistance)
1278
+ continue;
1279
+ const distance = levenshteinDistance(target, candidate);
1280
+ const similarity = stringSimilarity(target, candidate);
1281
+ if ((similarity > maxSimilarity && similarity >= threshold) ||
1282
+ (similarity >= threshold && distance < minDistance)) {
1283
+ maxSimilarity = similarity;
1284
+ minDistance = distance;
1285
+ mostSimilar = candidate;
1286
+ if (distance <= 1 || similarity > 0.95)
1287
+ break;
1288
+ }
1289
+ }
1290
+ return mostSimilar;
1291
+ }
903
1292
  /**
904
1293
  * Cek apakah string mungkin merupakan variasi dari kata kotor
905
1294
  * menggunakan kesamaan string
@@ -950,34 +1339,156 @@ function clusterSimilarWords(words, threshold = 0.8) {
950
1339
  }
951
1340
  /**
952
1341
  * Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
1342
+ * dengan optimasi untuk mengurangi kompleksitas
953
1343
  *
954
1344
  * @param text Teks yang akan diperiksa
955
1345
  * @param profanityWords Daftar kata kotor
956
1346
  * @param threshold Batas minimum kesamaan (default: 0.8)
957
1347
  * @returns Array kata yang mungkin merupakan kata kotor
958
1348
  */
1349
+ /**
1350
+ * Cari kata-kata kotor yang mungkin dari teks berdasarkan kemiripan string
1351
+ * dengan optimasi biar prosesnya nggak terlalu berat
1352
+ *
1353
+ * @param text Teks yang mau dicek
1354
+ * @param profanityWords Daftar kata-kata kotor/kasar
1355
+ * @param threshold Batas minimal kemiripan (default: 0.8)
1356
+ * @returns Array kata yang kemungkinan kata kotor/kasar
1357
+ */
959
1358
  function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.8) {
960
1359
  const result = [];
961
- // Pisahkan teks menjadi kata-kata
1360
+ // Optimasi dengan bikin map kata-kata kotor berdasarkan huruf pertama
1361
+ const profanityMap = new Map();
1362
+ // Kelompokkan kata-kata kotor berdasarkan huruf pertama biar pencariannya lebih cepat
1363
+ for (const word of profanityWords) {
1364
+ if (word.length < 1)
1365
+ continue;
1366
+ const firstChar = word[0].toLowerCase();
1367
+ if (!profanityMap.has(firstChar)) {
1368
+ profanityMap.set(firstChar, []);
1369
+ }
1370
+ profanityMap.get(firstChar).push(word);
1371
+ }
962
1372
  const words = text.toLowerCase().split(/\s+/);
963
1373
  for (const word of words) {
964
- // Lewati kata-kata yang terlalu pendek
965
1374
  if (word.length < 3)
966
1375
  continue;
967
- for (const profanity of profanityWords) {
1376
+ // Cuma bandingin dengan kata-kata kotor yang huruf pertamanya sama
1377
+ // atau yang perbedaan panjangnya masih masuk akal
1378
+ const firstChar = word[0];
1379
+ const candidateWords = profanityMap.get(firstChar) || [];
1380
+ // Cek juga huruf-huruf yang berdekatan (buat antisipasi typo di huruf pertama)
1381
+ // Ini opsional tapi bikin deteksinya lebih bagus
1382
+ const charCode = firstChar.charCodeAt(0);
1383
+ const prevChar = String.fromCharCode(charCode - 1);
1384
+ const nextChar = String.fromCharCode(charCode + 1);
1385
+ const adjacentCandidates = [
1386
+ ...(profanityMap.get(prevChar) || []),
1387
+ ...(profanityMap.get(nextChar) || []),
1388
+ ];
1389
+ // Gabungin kandidat-kandidatnya, prioritasin yang huruf pertamanya sama persis
1390
+ const allCandidates = [...candidateWords, ...adjacentCandidates];
1391
+ // Filter kandidat berdasarkan perbedaan panjang sebelum ngitung kemiripannya
1392
+ const lengthFilteredCandidates = allCandidates.filter((candidate) => Math.abs(candidate.length - word.length) <= 2);
1393
+ // Cari yang paling cocok
1394
+ let bestMatch = null;
1395
+ for (const profanity of lengthFilteredCandidates) {
968
1396
  const similarity = stringSimilarity(word, profanity);
969
- if (similarity >= threshold) {
970
- result.push({
1397
+ if (similarity >= threshold &&
1398
+ (!bestMatch || similarity > bestMatch.similarity)) {
1399
+ bestMatch = {
971
1400
  word,
972
1401
  original: profanity,
973
1402
  similarity,
974
- });
975
- break;
1403
+ };
976
1404
  }
977
1405
  }
1406
+ if (bestMatch) {
1407
+ result.push(bestMatch);
1408
+ }
978
1409
  }
979
1410
  return result;
980
1411
  }
1412
+ /**
1413
+ * Cari kata-kata kotor yang mungkin dari teks menggunakan Levenshtein distance
1414
+ *
1415
+ * @param text Teks yang akan diperiksa
1416
+ * @param profanityWords Daftar kata kotor
1417
+ * @param threshold Batas minimum kesamaan (default: 0.8)
1418
+ * @param maxDistance Jarak Levenshtein maksimal (default: 2)
1419
+ * @returns Array kata yang mungkin merupakan kata kotor
1420
+ */
1421
+ function findProfanityByLevenshteinDistance(text, profanityWords, threshold = 0.8, maxDistance = 2) {
1422
+ const result = [];
1423
+ // map kata-kata kotor dikelompokkan sesuai panjangnya
1424
+ const profanityByLength = new Map();
1425
+ // Kelompokkin kata-kata kotor berdasarkan panjangnya biar pencarian lebih cepat
1426
+ for (const word of profanityWords) {
1427
+ const length = word.length;
1428
+ if (!profanityByLength.has(length)) {
1429
+ profanityByLength.set(length, []);
1430
+ }
1431
+ profanityByLength.get(length).push(word);
1432
+ }
1433
+ const words = text.toLowerCase().split(/\s+/);
1434
+ for (const word of words) {
1435
+ if (word.length < 3)
1436
+ continue;
1437
+ let bestMatch = null;
1438
+ for (let len = Math.max(3, word.length - maxDistance); len <= word.length + maxDistance; len++) {
1439
+ const candidates = profanityByLength.get(len) || [];
1440
+ for (const profanity of candidates) {
1441
+ if (!isCharacterCountSimilar(word, profanity, maxDistance)) {
1442
+ continue;
1443
+ }
1444
+ const distance = levenshteinDistance(word, profanity);
1445
+ if (distance <= maxDistance) {
1446
+ const similarity = 1 - distance / Math.max(word.length, profanity.length);
1447
+ if (similarity >= threshold &&
1448
+ (!bestMatch || similarity > bestMatch.similarity)) {
1449
+ bestMatch = {
1450
+ word,
1451
+ original: profanity,
1452
+ similarity,
1453
+ distance,
1454
+ };
1455
+ if (distance === 0 || similarity > 0.95) {
1456
+ break;
1457
+ }
1458
+ }
1459
+ }
1460
+ }
1461
+ }
1462
+ if (bestMatch) {
1463
+ result.push(bestMatch);
1464
+ }
1465
+ }
1466
+ return result;
1467
+ }
1468
+ /**
1469
+ * Helper function to efficiently check if character counts between two strings
1470
+ * are similar enough to warrant a full Levenshtein calculation
1471
+ */
1472
+ function isCharacterCountSimilar(str1, str2, maxDifference) {
1473
+ const charCount1 = {};
1474
+ const charCount2 = {};
1475
+ for (const char of str1) {
1476
+ charCount1[char] = (charCount1[char] || 0) + 1;
1477
+ }
1478
+ for (const char of str2) {
1479
+ charCount2[char] = (charCount2[char] || 0) + 1;
1480
+ }
1481
+ let diffCount = 0;
1482
+ for (const char in charCount1) {
1483
+ diffCount += Math.abs((charCount1[char] || 0) - (charCount2[char] || 0));
1484
+ }
1485
+ for (const char in charCount2) {
1486
+ if (!charCount1[char]) {
1487
+ diffCount += charCount2[char];
1488
+ }
1489
+ }
1490
+ return diffCount <= maxDifference * 2;
1491
+ }
981
1492
 
982
1493
  const DEFAULT_OPTIONS = {
983
1494
  replaceWith: "*",
@@ -986,6 +1497,8 @@ const DEFAULT_OPTIONS = {
986
1497
  checkSubstring: false,
987
1498
  whitelist: [],
988
1499
  severityThreshold: 0,
1500
+ useLevenshtein: false,
1501
+ maxLevenshteinDistance: 2,
989
1502
  };
990
1503
  const FILTER_PRESETS = {
991
1504
  strict: {
@@ -1130,44 +1643,206 @@ function makeRandomGrawlixString(length) {
1130
1643
  return result;
1131
1644
  }
1132
1645
 
1133
- function findProfanity(text, options = {}) {
1134
- const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, } = { ...DEFAULT_OPTIONS, ...options };
1135
- const normalizedText = normalizeText(text);
1136
- let wordsToCheck = wordList.length > 0 ? wordList : [];
1137
- if (wordsToCheck.length === 0) {
1138
- if (categories || regions || severityThreshold > 0) {
1139
- wordsToCheck = wordObjects
1140
- .filter((word) => {
1141
- const matchCategory = categories
1142
- ? categories.includes(word.category)
1143
- : true;
1144
- const matchRegion = regions ? regions.includes(word.region) : true;
1145
- const matchSeverity = word.severity >= severityThreshold;
1146
- return matchCategory && matchRegion && matchSeverity;
1147
- })
1148
- .map((word) => word.word);
1646
+ /**
1647
+ * Implementasi algoritma Aho-Corasick untuk pencocokan string
1648
+ * Algoritma ini memungkinkan pencarian banyak pola dalam satu kali pembacaan teks
1649
+ */
1650
+ class AhoCorasick {
1651
+ constructor() {
1652
+ this.built = false;
1653
+ this.root = {
1654
+ children: new Map(),
1655
+ fail: null,
1656
+ output: new Set(),
1657
+ depth: 0,
1658
+ };
1659
+ }
1660
+ /**
1661
+ * Menambahkan pola ke dalam trie
1662
+ * @param pattern Pola yang akan ditambahkan
1663
+ */
1664
+ addPattern(pattern) {
1665
+ if (this.built) {
1666
+ throw new Error("Cannot add patterns after the automaton is built");
1149
1667
  }
1150
- else {
1151
- wordsToCheck = wordObjects.map((word) => word.word);
1668
+ let node = this.root;
1669
+ const normalizedPattern = pattern.toLowerCase();
1670
+ for (let i = 0; i < normalizedPattern.length; i++) {
1671
+ const char = normalizedPattern[i];
1672
+ if (!node.children.has(char)) {
1673
+ node.children.set(char, {
1674
+ children: new Map(),
1675
+ fail: null,
1676
+ output: new Set(),
1677
+ depth: node.depth + 1,
1678
+ char,
1679
+ });
1680
+ }
1681
+ node = node.children.get(char);
1682
+ }
1683
+ node.output.add(normalizedPattern);
1684
+ }
1685
+ /**
1686
+ * Membangun fungsi failure
1687
+ */
1688
+ build() {
1689
+ if (this.built)
1690
+ return;
1691
+ const queue = [];
1692
+ // Set fail pointer for depth 1 nodes to root
1693
+ for (const child of this.root.children.values()) {
1694
+ child.fail = this.root;
1695
+ queue.push(child);
1696
+ }
1697
+ // BFS to build failure links
1698
+ while (queue.length > 0) {
1699
+ const current = queue.shift();
1700
+ for (const [char, child] of current.children.entries()) {
1701
+ queue.push(child);
1702
+ let failNode = current.fail;
1703
+ // Find the longest proper suffix that is also a prefix
1704
+ while (failNode !== null && !failNode.children.has(char)) {
1705
+ failNode = failNode.fail;
1706
+ }
1707
+ if (failNode === null) {
1708
+ child.fail = this.root;
1709
+ }
1710
+ else {
1711
+ child.fail = failNode.children.get(char);
1712
+ // Add outputs from the fail state to this node
1713
+ for (const output of child.fail.output) {
1714
+ child.output.add(output);
1715
+ }
1716
+ }
1717
+ }
1152
1718
  }
1719
+ this.built = true;
1153
1720
  }
1154
- wordsToCheck = wordsToCheck.filter((word) => !whitelist.includes(word.toLocaleLowerCase()));
1721
+ /**
1722
+ * Mencari semua kemunculan pola dalam teks
1723
+ * @param text Teks yang akan dicari
1724
+ * @returns Map pola yang ditemukan dengan jumlah kemunculannya
1725
+ */
1726
+ search(text) {
1727
+ if (!this.built) {
1728
+ this.build();
1729
+ }
1730
+ const matches = new Map();
1731
+ const normalizedText = text.toLowerCase();
1732
+ let node = this.root;
1733
+ for (let i = 0; i < normalizedText.length; i++) {
1734
+ const char = normalizedText[i];
1735
+ // Follow failure links until we find a matching transition or reach root
1736
+ while (node !== this.root && !node.children.has(char)) {
1737
+ node = node.fail;
1738
+ }
1739
+ // Try to follow the transition
1740
+ if (node.children.has(char)) {
1741
+ node = node.children.get(char);
1742
+ }
1743
+ // Check for any matches at this node
1744
+ for (const match of node.output) {
1745
+ matches.set(match, (matches.get(match) || 0) + 1);
1746
+ }
1747
+ }
1748
+ return matches;
1749
+ }
1750
+ /**
1751
+ * Mencari semua kemunculan pola dalam teks dan mengembalikan hanya pola unik
1752
+ * @param text Teks yang akan dicari
1753
+ * @returns Set pola yang ditemukan
1754
+ */
1755
+ searchUnique(text) {
1756
+ const matches = this.search(text);
1757
+ return new Set(matches.keys());
1758
+ }
1759
+ /**
1760
+ * Mengecek apakah teks mengandung setidaknya satu pola
1761
+ * @param text Teks yang akan dicari
1762
+ * @returns Boolean apakah pola ditemukan
1763
+ */
1764
+ containsAny(text) {
1765
+ if (!this.built) {
1766
+ this.build();
1767
+ }
1768
+ const normalizedText = text.toLowerCase();
1769
+ let node = this.root;
1770
+ for (let i = 0; i < normalizedText.length; i++) {
1771
+ const char = normalizedText[i];
1772
+ while (node !== this.root && !node.children.has(char)) {
1773
+ node = node.fail;
1774
+ }
1775
+ if (node.children.has(char)) {
1776
+ node = node.children.get(char);
1777
+ }
1778
+ if (node.output.size > 0) {
1779
+ return true;
1780
+ }
1781
+ }
1782
+ return false;
1783
+ }
1784
+ }
1785
+
1786
+ const globalAhoCorasick = new AhoCorasick();
1787
+ let ahoCorasickInitialized = false;
1788
+ function initializeAhoCorasick(words) {
1789
+ if (ahoCorasickInitialized)
1790
+ return;
1791
+ for (const word of words) {
1792
+ globalAhoCorasick.addPattern(word);
1793
+ }
1794
+ globalAhoCorasick.build();
1795
+ ahoCorasickInitialized = true;
1796
+ }
1797
+ function findProfanity(text, options = {}) {
1798
+ const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, useLevenshtein = false, maxLevenshteinDistance = 2, } = { ...DEFAULT_OPTIONS, ...options };
1799
+ const normalizedText = normalizeText(text);
1800
+ let baseWordsToCheck = wordList.length > 0 ? wordList : [];
1801
+ if (baseWordsToCheck.length === 0) {
1802
+ const filteredWords = wordObjects.filter((word) => {
1803
+ const matchCategory = categories
1804
+ ? categories.includes(word.category)
1805
+ : true;
1806
+ const matchRegion = regions ? regions.includes(word.region) : true;
1807
+ const matchSeverity = word.severity >= severityThreshold;
1808
+ return matchCategory && matchRegion && matchSeverity;
1809
+ });
1810
+ baseWordsToCheck = filteredWords.map((word) => word.word);
1811
+ }
1812
+ const aliasMap = new Map();
1813
+ wordObjects.forEach((wordObj) => {
1814
+ if (wordObj.aliases && wordObj.aliases.length > 0) {
1815
+ const matchCategory = categories
1816
+ ? categories.includes(wordObj.category)
1817
+ : true;
1818
+ const matchRegion = regions ? regions.includes(wordObj.region) : true;
1819
+ const matchSeverity = wordObj.severity >= severityThreshold;
1820
+ if (matchCategory && matchRegion && matchSeverity) {
1821
+ wordObj.aliases.forEach((alias) => {
1822
+ aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
1823
+ });
1824
+ }
1825
+ }
1826
+ });
1827
+ const wordsToCheck = [
1828
+ ...baseWordsToCheck,
1829
+ ...Array.from(aliasMap.keys()),
1830
+ ].filter((word) => !whitelist.includes(word.toLowerCase()));
1155
1831
  if (wordsToCheck.length === 0) {
1156
1832
  return [];
1157
1833
  }
1158
1834
  const matches = new Set();
1159
- wordsToCheck.forEach((word) => {
1160
- const regex = createWordRegex(word, {
1161
- wholeWord: !checkSubstring,
1162
- caseSensitive: false,
1163
- leetSpeak: false,
1164
- detectSplit: false,
1165
- indonesianVariation: false,
1166
- });
1167
- while ((regex.exec(normalizedText)) !== null) {
1168
- matches.add(word.toLowerCase());
1835
+ const actualMatches = new Map();
1836
+ initializeAhoCorasick(wordsToCheck);
1837
+ const basicMatches = globalAhoCorasick.searchUnique(normalizedText);
1838
+ for (const match of basicMatches) {
1839
+ const originalWord = aliasMap.get(match.toLowerCase()) || match.toLowerCase();
1840
+ matches.add(originalWord);
1841
+ if (!actualMatches.has(originalWord)) {
1842
+ actualMatches.set(originalWord, []);
1169
1843
  }
1170
- });
1844
+ actualMatches.get(originalWord)?.push(match);
1845
+ }
1171
1846
  if (detectLeetSpeak) {
1172
1847
  wordsToCheck.forEach((word) => {
1173
1848
  const leetRegex = createWordRegex(word, {
@@ -1177,8 +1852,14 @@ function findProfanity(text, options = {}) {
1177
1852
  detectSplit: false,
1178
1853
  indonesianVariation: false,
1179
1854
  });
1180
- while ((leetRegex.exec(text)) !== null) {
1181
- matches.add(word.toLowerCase());
1855
+ let match;
1856
+ while ((match = leetRegex.exec(text)) !== null) {
1857
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1858
+ matches.add(originalWord);
1859
+ if (!actualMatches.has(originalWord)) {
1860
+ actualMatches.set(originalWord, []);
1861
+ }
1862
+ actualMatches.get(originalWord)?.push(match[0]);
1182
1863
  }
1183
1864
  });
1184
1865
  }
@@ -1191,33 +1872,64 @@ function findProfanity(text, options = {}) {
1191
1872
  detectSplit: false,
1192
1873
  indonesianVariation: true,
1193
1874
  });
1194
- while ((variantRegex.exec(text)) !== null) {
1195
- matches.add(word.toLowerCase());
1875
+ let match;
1876
+ while ((match = variantRegex.exec(text)) !== null) {
1877
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1878
+ matches.add(originalWord);
1879
+ if (!actualMatches.has(originalWord)) {
1880
+ actualMatches.set(originalWord, []);
1881
+ }
1882
+ actualMatches.get(originalWord)?.push(match[0]);
1196
1883
  }
1197
1884
  });
1198
1885
  }
1199
1886
  if (detectSplit) {
1200
- if (detectSplitWords(text, wordsToCheck)) {
1201
- wordsToCheck.forEach((word) => {
1202
- const splitRegex = createWordRegex(word, {
1203
- wholeWord: false,
1204
- caseSensitive: false,
1205
- leetSpeak: false,
1206
- detectSplit: true,
1207
- indonesianVariation: false,
1208
- });
1209
- if (splitRegex.test(text)) {
1210
- matches.add(word.toLowerCase());
1211
- }
1887
+ wordsToCheck.forEach((word) => {
1888
+ const splitRegex = createWordRegex(word, {
1889
+ wholeWord: false,
1890
+ caseSensitive: false,
1891
+ leetSpeak: false,
1892
+ detectSplit: true,
1893
+ indonesianVariation: false,
1212
1894
  });
1213
- }
1895
+ let match;
1896
+ while ((match = splitRegex.exec(text)) !== null) {
1897
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1898
+ matches.add(originalWord);
1899
+ if (!actualMatches.has(originalWord)) {
1900
+ actualMatches.set(originalWord, []);
1901
+ }
1902
+ actualMatches.get(originalWord)?.push(match[0]);
1903
+ }
1904
+ });
1214
1905
  }
1215
1906
  if (detectSimilarity) {
1216
- const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
1217
- possibleProfanity.forEach((item) => {
1218
- matches.add(item.original.toLowerCase());
1219
- });
1907
+ if (useLevenshtein) {
1908
+ const possibleProfanity = findProfanityByLevenshteinDistance(text, wordsToCheck, similarityThreshold, maxLevenshteinDistance);
1909
+ possibleProfanity.forEach((item) => {
1910
+ const originalWord = aliasMap.get(item.original.toLowerCase()) ||
1911
+ item.original.toLowerCase();
1912
+ matches.add(originalWord);
1913
+ if (!actualMatches.has(originalWord)) {
1914
+ actualMatches.set(originalWord, []);
1915
+ }
1916
+ actualMatches.get(originalWord)?.push(item.word);
1917
+ });
1918
+ }
1919
+ else {
1920
+ const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
1921
+ possibleProfanity.forEach((item) => {
1922
+ matches.add(item.original.toLowerCase());
1923
+ const originalWord = aliasMap.get(item.original.toLowerCase()) ||
1924
+ item.original.toLowerCase();
1925
+ if (!actualMatches.has(originalWord)) {
1926
+ actualMatches.set(originalWord, []);
1927
+ }
1928
+ actualMatches.get(originalWord)?.push(item.word);
1929
+ });
1930
+ }
1220
1931
  }
1932
+ findProfanity.lastActualMatches = actualMatches;
1221
1933
  return Array.from(matches);
1222
1934
  }
1223
1935
  /**
@@ -1307,14 +2019,20 @@ function calculateSeverity(matchDetails) {
1307
2019
  * @returns FilterResult dengan hasil filter
1308
2020
  */
1309
2021
  function filter(text, options = {}) {
1310
- const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, } = { ...DEFAULT_OPTIONS, ...options };
2022
+ const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, detectSplit = false, detectSimilarity = false, useLevenshtein = false, maxLevenshteinDistance = 2, similarityThreshold = 0.8, } = { ...DEFAULT_OPTIONS, ...options };
1311
2023
  const matches = findProfanity(text, {
1312
2024
  ...options,
1313
2025
  detectLeetSpeak,
1314
2026
  whitelist,
1315
2027
  checkSubstring,
1316
2028
  indonesianVariation,
2029
+ detectSplit,
2030
+ detectSimilarity,
2031
+ useLevenshtein,
2032
+ maxLevenshteinDistance,
2033
+ similarityThreshold,
1317
2034
  });
2035
+ const actualMatches = findProfanity.lastActualMatches || new Map();
1318
2036
  const matchDetails = findProfanityWithMetadata(text, options);
1319
2037
  if (matches.length === 0) {
1320
2038
  return {
@@ -1329,36 +2047,120 @@ function filter(text, options = {}) {
1329
2047
  const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
1330
2048
  (m.aliases &&
1331
2049
  m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
1332
- const regex = createWordRegex(word, {
1333
- wholeWord: true,
1334
- caseSensitive: false,
1335
- leetSpeak: false,
1336
- detectSplit: false,
1337
- indonesianVariation: false,
2050
+ const variants = actualMatches.get(word.toLowerCase()) || [];
2051
+ variants.push(word);
2052
+ const uniqueVariants = [...new Set(variants)];
2053
+ uniqueVariants.forEach((variant) => {
2054
+ const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
2055
+ let match;
2056
+ while ((match = regex.exec(filteredText)) !== null) {
2057
+ const originalWord = match[0];
2058
+ if (whitelist.includes(originalWord.toLowerCase()))
2059
+ continue;
2060
+ let censoredWord;
2061
+ if (useRandomGrawlix) {
2062
+ censoredWord = makeRandomGrawlixString(originalWord.length);
2063
+ }
2064
+ else {
2065
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2066
+ }
2067
+ replacements.push({
2068
+ original: originalWord,
2069
+ censored: censoredWord,
2070
+ metadata,
2071
+ });
2072
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"), censoredWord);
2073
+ }
1338
2074
  });
1339
- let match;
1340
- const textToSearch = filteredText;
1341
- regex.lastIndex = 0;
1342
- while ((match = regex.exec(textToSearch)) !== null) {
1343
- const originalWord = match[0];
1344
- if (whitelist.includes(originalWord.toLowerCase()))
1345
- continue;
1346
- let censoredWord;
1347
- if (useRandomGrawlix) {
1348
- censoredWord = makeRandomGrawlixString(originalWord.length);
2075
+ if (detectSplit || detectLeetSpeak) {
2076
+ if (detectLeetSpeak) {
2077
+ const leetRegex = createWordRegex(word, {
2078
+ wholeWord: true,
2079
+ caseSensitive: false,
2080
+ leetSpeak: true,
2081
+ detectSplit: false,
2082
+ indonesianVariation: false,
2083
+ });
2084
+ let match;
2085
+ while ((match = leetRegex.exec(filteredText)) !== null) {
2086
+ const originalWord = match[0];
2087
+ if (whitelist.includes(originalWord.toLowerCase()))
2088
+ continue;
2089
+ let censoredWord;
2090
+ if (useRandomGrawlix) {
2091
+ censoredWord = makeRandomGrawlixString(originalWord.length);
2092
+ }
2093
+ else {
2094
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2095
+ }
2096
+ replacements.push({
2097
+ original: originalWord,
2098
+ censored: censoredWord,
2099
+ metadata,
2100
+ });
2101
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), "g"), censoredWord);
2102
+ }
1349
2103
  }
1350
- else {
1351
- censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2104
+ if (detectSplit) {
2105
+ const splitRegex = createWordRegex(word, {
2106
+ wholeWord: false,
2107
+ caseSensitive: false,
2108
+ leetSpeak: false,
2109
+ detectSplit: true,
2110
+ indonesianVariation: false,
2111
+ });
2112
+ let match;
2113
+ while ((match = splitRegex.exec(filteredText)) !== null) {
2114
+ const originalWord = match[0];
2115
+ if (whitelist.includes(originalWord.toLowerCase()))
2116
+ continue;
2117
+ let censoredWord;
2118
+ if (useRandomGrawlix) {
2119
+ censoredWord = makeRandomGrawlixString(originalWord.length);
2120
+ }
2121
+ else {
2122
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2123
+ }
2124
+ replacements.push({
2125
+ original: originalWord,
2126
+ censored: censoredWord,
2127
+ metadata,
2128
+ });
2129
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), "g"), censoredWord);
2130
+ }
1352
2131
  }
1353
- replacements.push({
1354
- original: originalWord,
1355
- censored: censoredWord,
1356
- metadata,
1357
- });
1358
- const replaceRegex = new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g");
1359
- filteredText = filteredText.replace(replaceRegex, censoredWord);
1360
2132
  }
1361
2133
  });
2134
+ if (detectSimilarity && useLevenshtein) {
2135
+ matches.forEach((word) => {
2136
+ const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
2137
+ (m.aliases &&
2138
+ m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
2139
+ const variants = actualMatches.get(word.toLowerCase()) || [];
2140
+ variants.forEach((variant) => {
2141
+ const exactVariantRegex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
2142
+ let match;
2143
+ while ((match = exactVariantRegex.exec(filteredText)) !== null) {
2144
+ const originalWord = match[0];
2145
+ if (whitelist.includes(originalWord.toLowerCase()))
2146
+ continue;
2147
+ let censoredWord;
2148
+ if (useRandomGrawlix) {
2149
+ censoredWord = makeRandomGrawlixString(originalWord.length);
2150
+ }
2151
+ else {
2152
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2153
+ }
2154
+ replacements.push({
2155
+ original: originalWord,
2156
+ censored: censoredWord,
2157
+ metadata,
2158
+ });
2159
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"), censoredWord);
2160
+ }
2161
+ });
2162
+ });
2163
+ }
1362
2164
  return {
1363
2165
  filtered: filteredText,
1364
2166
  censored: replacements.length,
@@ -1403,8 +2205,20 @@ function analyze(text, options = {}) {
1403
2205
  const severityScore = calculateSeverity(matchDetails);
1404
2206
  let similarWords = [];
1405
2207
  if (mergedOptions.detectSimilarity) {
1406
- const wordList = matchDetails.map((word) => word.word);
1407
- similarWords = findPossibleProfanityBySimiliarity(text, wordList, mergedOptions.similarityThreshold || 0.8);
2208
+ if (matchDetails.length > 0) {
2209
+ const wordList = matchDetails.map((word) => word.word);
2210
+ if (mergedOptions.useLevenshtein) {
2211
+ const levenshteinResults = findProfanityByLevenshteinDistance(text, wordList, mergedOptions.similarityThreshold || 0.8, mergedOptions.maxLevenshteinDistance || 2);
2212
+ similarWords = levenshteinResults.map((item) => ({
2213
+ word: item.word,
2214
+ original: item.original,
2215
+ similarity: item.similarity,
2216
+ }));
2217
+ }
2218
+ else {
2219
+ similarWords = findPossibleProfanityBySimiliarity(text, wordList, mergedOptions.similarityThreshold || 0.8);
2220
+ }
2221
+ }
1408
2222
  }
1409
2223
  return {
1410
2224
  hasProfanity: true,
@@ -1633,10 +2447,25 @@ class IDProfanityFilter {
1633
2447
  /**
1634
2448
  * Mengaktifkan deteksi berdasarkan kesamaan
1635
2449
  * @param threshold Threshold kesamaan (0-1)
2450
+ * @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
2451
+ * @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
2452
+ */
2453
+ enableSimilarityDetection(threshold = 0.8, useLevenshtein = false, maxLevenshteinDistance = 2) {
2454
+ this.options.detectSimilarity = true;
2455
+ this.options.similarityThreshold = threshold;
2456
+ this.options.useLevenshtein = useLevenshtein;
2457
+ this.options.maxLevenshteinDistance = maxLevenshteinDistance;
2458
+ }
2459
+ /**
2460
+ * Mengaktifkan deteksi berbasis Levenshtein distance
2461
+ * @param threshold Threshold kesamaan (0-1)
2462
+ * @param maxDistance Jarak maksimal Levenshtein (default: 2)
1636
2463
  */
1637
- enableSimilarityDetection(threshold = 0.8) {
2464
+ enableLevenshteinDetection(threshold = 0.8, maxDistance = 2) {
1638
2465
  this.options.detectSimilarity = true;
2466
+ this.options.useLevenshtein = true;
1639
2467
  this.options.similarityThreshold = threshold;
2468
+ this.options.maxLevenshteinDistance = maxDistance;
1640
2469
  }
1641
2470
  }
1642
2471
  const idFilter = {
@@ -1682,8 +2511,10 @@ exports.escapeRegExp = escapeRegExp;
1682
2511
  exports.filter = filter;
1683
2512
  exports.findCategories = findCategories;
1684
2513
  exports.findMostSimilar = findMostSimilar;
2514
+ exports.findMostSimilarWithLevenshtein = findMostSimilarWithLevenshtein;
1685
2515
  exports.findPossibleProfanityBySimiliarity = findPossibleProfanityBySimiliarity;
1686
2516
  exports.findProfanity = findProfanity;
2517
+ exports.findProfanityByLevenshteinDistance = findProfanityByLevenshteinDistance;
1687
2518
  exports.findProfanityWithMetadata = findProfanityWithMetadata;
1688
2519
  exports.findRegions = findRegions;
1689
2520
  exports.getContextAroundIndex = getContextAroundIndex;