@sideid/id-profanity-filter 1.10.6 → 1.11.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/.eslintrc.js +44 -44
  2. package/.github/workflows/release.yml +62 -0
  3. package/CONTRIBUTING.md +150 -150
  4. package/LICENSE +21 -21
  5. package/README.md +548 -548
  6. package/dist/index.d.ts +989 -0
  7. package/dist/index.esm.js +545 -50
  8. package/dist/index.esm.js.map +1 -1
  9. package/dist/index.js +545 -50
  10. package/dist/index.js.map +1 -1
  11. package/dist/types/constants/categories/blasphemy.d.ts +4 -0
  12. package/dist/types/constants/categories/disgusting.d.ts +4 -0
  13. package/dist/types/constants/categories/drugs.d.ts +4 -0
  14. package/dist/types/constants/categories/profanity.d.ts +4 -0
  15. package/dist/types/constants/categories/slur.d.ts +4 -0
  16. package/dist/{constants → types/constants}/wordList.d.ts +1 -1
  17. package/dist/{core → types/core}/analyzer.d.ts +1 -1
  18. package/dist/{core → types/core}/filter.d.ts +1 -1
  19. package/dist/{core → types/core}/matcher.d.ts +1 -1
  20. package/dist/types/index.d.ts +375 -57
  21. package/dist/types/types/index.d.ts +59 -0
  22. package/dist/types/utils/ahoCorasick.d.ts +36 -0
  23. package/dist/{utils → types/utils}/similarityUtils.d.ts +10 -0
  24. package/eslint.config.mjs +40 -40
  25. package/examples/advanced.ts +120 -120
  26. package/examples/basic.ts +71 -71
  27. package/examples/custom-list.ts +140 -140
  28. package/jest.config.mjs +10 -10
  29. package/package.json +2 -1
  30. package/prettierrc +6 -6
  31. package/rollup.config.mjs +40 -35
  32. package/src/constants/categories/blasphemy.ts +25 -0
  33. package/src/constants/categories/disgusting.ts +82 -0
  34. package/src/constants/categories/drugs.ts +72 -0
  35. package/src/constants/categories/profanity.ts +139 -0
  36. package/src/constants/categories/slur.ts +102 -0
  37. package/src/constants/regions/general.ts +9 -0
  38. package/src/constants/regions/jawa.ts +356 -354
  39. package/src/core/analyzer.ts +28 -7
  40. package/src/core/matcher.ts +25 -18
  41. package/src/index.ts +15 -15
  42. package/src/utils/ahoCorasick.ts +179 -0
  43. package/src/utils/similarityUtils.ts +157 -20
  44. package/tsconfig.json +115 -115
  45. package/.github/workflows/ci.yml +0 -0
  46. package/dist/constants/categories/index.d.ts +0 -9
  47. package/dist/constants/regions/index.d.ts +0 -8
  48. package/test.js +0 -184
  49. /package/dist/{config → types/config}/options.d.ts +0 -0
  50. /package/dist/{constants → types/constants}/categories/insult.d.ts +0 -0
  51. /package/dist/{constants → types/constants}/categories/sexual.d.ts +0 -0
  52. /package/dist/{constants → types/constants}/regions/batak.d.ts +0 -0
  53. /package/dist/{constants → types/constants}/regions/betawi.d.ts +0 -0
  54. /package/dist/{constants → types/constants}/regions/general.d.ts +0 -0
  55. /package/dist/{constants → types/constants}/regions/jawa.d.ts +0 -0
  56. /package/dist/{constants → types/constants}/regions/sunda.d.ts +0 -0
  57. /package/dist/{utils → types/utils}/regexUtils.d.ts +0 -0
  58. /package/dist/{utils → types/utils}/stringUtils.d.ts +0 -0
package/dist/index.js CHANGED
@@ -201,6 +201,15 @@ const general = [
201
201
  description: "Simbol yang sering digunakan untuk menandai konten pornografi",
202
202
  context: "Digunakan untuk menandai konten seksual eksplisit",
203
203
  },
204
+ {
205
+ word: "xnxx",
206
+ category: "sexual",
207
+ region: "general",
208
+ severity: 0.6,
209
+ aliases: ["xnxx", "xnx"],
210
+ description: "simbol yang sering digunakan untuk menandai konten pornografi",
211
+ context: "Digunakan untuk menandai konten seksual eksplisit",
212
+ }
204
213
  ];
205
214
  general.map((item) => item.word);
206
215
 
@@ -218,7 +227,7 @@ const jawa = [
218
227
  word: "jancok",
219
228
  category: "sexual",
220
229
  region: "jawa",
221
- severity: 0.8,
230
+ severity: 0.9,
222
231
  aliases: ["jancuk", "jncok", "jancuk", "jncuk", "dancok", "dancuk"],
223
232
  description: "Kata umpatan kasar dalam Bahasa Jawa",
224
233
  context: "Umpatan kasar yang umum digunakan di Jawa Timur",
@@ -254,7 +263,7 @@ const jawa = [
254
263
  word: "mbokne ancok",
255
264
  category: "insult",
256
265
  region: "jawa",
257
- severity: 0.8,
266
+ severity: 0.9,
258
267
  aliases: ["mbokne", "mbokneancok"],
259
268
  description: "Umpatan yang menyinggung ibu seseorang",
260
269
  context: "Umpatan kasar yang menyinggung orangtua orang lain",
@@ -263,7 +272,7 @@ const jawa = [
263
272
  word: "pekok",
264
273
  category: "insult",
265
274
  region: "jawa",
266
- severity: 0.6,
275
+ severity: 0.7,
267
276
  aliases: ["pekak", "pekilk"],
268
277
  description: "Kata hinaan yang menunjukkan kebodohan",
269
278
  context: "Hinaan untuk menyebut orang yang dianggap sangat bodoh",
@@ -295,6 +304,248 @@ const jawa = [
295
304
  description: "Secara harfiah berarti kepalamu, digunakan sebagai umpatan ringan",
296
305
  context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
297
306
  },
307
+ {
308
+ word: "kontol",
309
+ category: "sexual",
310
+ region: "jawa",
311
+ severity: 0.8,
312
+ aliases: ["kntl", "kontl"],
313
+ description: "Mengacu ke alat kelamin laki-laki",
314
+ context: "Kata vulgar yang sering digunakan sebagai umpatan kasar",
315
+ },
316
+ {
317
+ word: "tempek",
318
+ category: "sexual",
319
+ region: "jawa",
320
+ severity: 0.8,
321
+ aliases: ["mpek", "torok", "tempk"],
322
+ description: "Mengacu pada alat kelamin perempuan",
323
+ context: "Kata vulgar yang digunakan sebagai umpatan atau hinaan",
324
+ },
325
+ {
326
+ word: "silit",
327
+ category: "insult",
328
+ region: "jawa",
329
+ severity: 0.6,
330
+ aliases: ["selet", "tilis"],
331
+ description: "Mengacu pada bagian dubur atau anus",
332
+ context: "Kata kasar yang digunakan sebagai hinaan",
333
+ },
334
+ {
335
+ word: "mbahmu",
336
+ category: "insult",
337
+ region: "jawa",
338
+ severity: 0.6,
339
+ aliases: ["mbahmu kiper", "mbah mu"],
340
+ description: "Hinaan yang menyinggung nenek/kakek seseorang",
341
+ context: "Umpatan ringan untuk menanggapi sesuatu yang tidak masuk akal",
342
+ },
343
+ {
344
+ word: "makmu",
345
+ category: "insult",
346
+ region: "jawa",
347
+ severity: 0.7,
348
+ aliases: ["mak mu", "mamamu"],
349
+ description: "Hinaan yang menyinggung ibu seseorang",
350
+ context: "Umpatan yang dianggap kasar karena menyinggung orang tua",
351
+ },
352
+ {
353
+ word: "bajingan",
354
+ category: "insult",
355
+ region: "jawa",
356
+ severity: 0.7,
357
+ aliases: [
358
+ "bajilak",
359
+ "bajhingan",
360
+ "bajingak",
361
+ "bajingseng",
362
+ "bajindul",
363
+ "bajigur",
364
+ "jingan",
365
+ ],
366
+ description: "Sebutan untuk orang yang dianggap jahat atau tidak bermoral",
367
+ context: "Umpatan untuk mengekspresikan kemarahan atau kekesalan",
368
+ },
369
+ {
370
+ word: "cocote",
371
+ category: "insult",
372
+ region: "jawa",
373
+ severity: 0.6,
374
+ aliases: ["cocot", "bacot", "nyocot"],
375
+ description: "Mengacu pada mulut dengan konotasi negatif",
376
+ context: "Umpatan untuk menyuruh seseorang berhenti berbicara",
377
+ },
378
+ {
379
+ word: "ngentot",
380
+ category: "sexual",
381
+ region: "jawa",
382
+ severity: 0.9,
383
+ aliases: ["kentu", "kentot", "iclik", "ngtt", "iclk"],
384
+ description: "Mengacu pada aktivitas seksual",
385
+ context: "Kata vulgar yang digunakan sebagai umpatan kasar",
386
+ },
387
+ {
388
+ word: "edan",
389
+ category: "insult",
390
+ region: "jawa",
391
+ severity: 0.5,
392
+ aliases: ["gendeng", "gila", "gendheng", "sarap"],
393
+ description: "Secara harfiah berarti gila atau tidak waras",
394
+ context: "Umpatan untuk menyebut seseorang yang dianggap tidak masuk akal",
395
+ },
396
+ {
397
+ word: "dapuranmu",
398
+ category: "insult",
399
+ region: "jawa",
400
+ severity: 0.6,
401
+ aliases: ["raimu", "rai mu"],
402
+ description: "Secara harfiah mengacu pada wajah atau rupa seseorang",
403
+ context: "Umpatan untuk menghina penampilan atau wajah seseorang",
404
+ },
405
+ {
406
+ word: "damput",
407
+ category: "insult",
408
+ region: "jawa",
409
+ severity: 0.7,
410
+ aliases: ["diamput"],
411
+ description: "Variasi bentuk umpatan dengan makna serupa dengan jancok",
412
+ context: "Umpatan kasar untuk mengekspresikan kemarahan",
413
+ },
414
+ {
415
+ word: "mbathang",
416
+ category: "insult",
417
+ region: "jawa",
418
+ severity: 0.7,
419
+ aliases: ["mbatang"],
420
+ description: "Secara harfiah berarti bangkai",
421
+ context: "Umpatan kasar untuk menghina seseorang",
422
+ },
423
+ {
424
+ word: "ndlogok",
425
+ category: "insult",
426
+ region: "jawa",
427
+ severity: 0.6,
428
+ aliases: ["ndelodok", "ndlodok"],
429
+ description: "Mengacu pada tindakan yang dianggap bodoh atau tidak masuk akal",
430
+ context: "Hinaan untuk mengkritik tindakan seseorang",
431
+ },
432
+ {
433
+ word: "nggateli",
434
+ category: "insult",
435
+ region: "jawa",
436
+ severity: 0.5,
437
+ aliases: ["gateli", "gathel"],
438
+ description: "Secara harfiah berarti gatal atau menyebalkan",
439
+ context: "Ungkapan untuk menunjukkan kekesalan terhadap perilaku seseorang",
440
+ },
441
+ {
442
+ word: "perek",
443
+ category: "sexual",
444
+ region: "jawa",
445
+ severity: 0.8,
446
+ aliases: ["lonthe", "pelacur"],
447
+ description: "Istilah merendahkan untuk pekerja seks komersial",
448
+ context: "Kata kasar untuk menghina wanita",
449
+ },
450
+ {
451
+ word: "picek",
452
+ category: "insult",
453
+ region: "jawa",
454
+ severity: 0.6,
455
+ aliases: ["pcek", "buta"],
456
+ description: "Secara harfiah berarti buta atau tidak bisa melihat",
457
+ context: "Hinaan untuk orang yang dianggap tidak bisa melihat kenyataan",
458
+ },
459
+ {
460
+ word: "untumu",
461
+ category: "insult",
462
+ region: "jawa",
463
+ severity: 0.5,
464
+ aliases: ["gigimu", "untu mu"],
465
+ description: "Secara harfiah berarti gigimu",
466
+ context: "Umpatan ringan untuk menanggapi sesuatu yang tidak disetujui",
467
+ },
468
+ {
469
+ word: "goblog",
470
+ category: "insult",
471
+ region: "jawa",
472
+ severity: 0.7,
473
+ aliases: ["ghoblog", "goblok", "gobhlok", "pekok"],
474
+ description: "Kata hinaan yang menunjukkan kebodohan ekstrem",
475
+ context: "Hinaan untuk menyebut seseorang yang dianggap sangat bodoh",
476
+ },
477
+ {
478
+ word: "tolol",
479
+ category: "insult",
480
+ region: "jawa",
481
+ severity: 0.7,
482
+ aliases: ["tholol", "tlol"],
483
+ description: "Kata hinaan yang menunjukkan kebodohan",
484
+ context: "Hinaan untuk menyebut seseorang yang dianggap bodoh",
485
+ },
486
+ {
487
+ word: "budheg",
488
+ category: "insult",
489
+ region: "jawa",
490
+ severity: 0.6,
491
+ aliases: ["budeg", "bdeg"],
492
+ description: "Secara harfiah berarti tuli atau tidak bisa mendengar",
493
+ context: "Hinaan untuk orang yang dianggap tidak mau mendengarkan",
494
+ },
495
+ {
496
+ word: "jiangkrik",
497
+ category: "insult",
498
+ region: "jawa",
499
+ severity: 0.4,
500
+ aliases: ["jiangkrek", "jangkrik"],
501
+ description: "Secara harfiah berarti jangkrik, digunakan sebagai eufemisme",
502
+ context: "Umpatan ringan sebagai pengganti kata kasar yang lebih vulgar",
503
+ },
504
+ {
505
+ word: "diamput",
506
+ category: "insult",
507
+ region: "jawa",
508
+ severity: 0.8,
509
+ aliases: ["damput", "djamput"],
510
+ description: "Bentuk umpatan kasar dengan makna serupa jancok",
511
+ context: "Kata kasar untuk mengekspresikan kemarahan",
512
+ },
513
+ {
514
+ word: "celeng",
515
+ category: "insult",
516
+ region: "jawa",
517
+ severity: 0.6,
518
+ aliases: ["cleng", "babi hutan"],
519
+ description: "Secara harfiah berarti babi hutan",
520
+ context: "Hinaan untuk orang yang dianggap jorok atau rakus",
521
+ },
522
+ {
523
+ word: "kampret",
524
+ category: "insult",
525
+ region: "jawa",
526
+ severity: 0.5,
527
+ aliases: ["kmpret", "kmprt"],
528
+ description: "Secara harfiah berarti kelelawar kecil",
529
+ context: "Umpatan ringan untuk mengekspresikan kekesalan",
530
+ },
531
+ {
532
+ word: "ndeso",
533
+ category: "insult",
534
+ region: "jawa",
535
+ severity: 0.4,
536
+ aliases: ["ndesa", "deso"],
537
+ description: "Secara harfiah berarti dari desa atau kampungan",
538
+ context: "Hinaan untuk orang yang dianggap kurang modern atau berpendidikan",
539
+ },
540
+ {
541
+ word: "kere",
542
+ category: "insult",
543
+ region: "jawa",
544
+ severity: 0.5,
545
+ aliases: ["miskin", "mlarat"],
546
+ description: "Secara harfiah berarti miskin atau tidak punya uang",
547
+ context: "Hinaan untuk status ekonomi seseorang yang dianggap rendah",
548
+ },
298
549
  {
299
550
  word: "itil",
300
551
  category: "sexual",
@@ -1088,29 +1339,73 @@ function clusterSimilarWords(words, threshold = 0.8) {
1088
1339
  }
1089
1340
  /**
1090
1341
  * Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
1342
+ * dengan optimasi untuk mengurangi kompleksitas
1091
1343
  *
1092
1344
  * @param text Teks yang akan diperiksa
1093
1345
  * @param profanityWords Daftar kata kotor
1094
1346
  * @param threshold Batas minimum kesamaan (default: 0.8)
1095
1347
  * @returns Array kata yang mungkin merupakan kata kotor
1096
1348
  */
1349
+ /**
1350
+ * Cari kata-kata kotor yang mungkin dari teks berdasarkan kemiripan string
1351
+ * dengan optimasi biar prosesnya nggak terlalu berat
1352
+ *
1353
+ * @param text Teks yang mau dicek
1354
+ * @param profanityWords Daftar kata-kata kotor/kasar
1355
+ * @param threshold Batas minimal kemiripan (default: 0.8)
1356
+ * @returns Array kata yang kemungkinan kata kotor/kasar
1357
+ */
1097
1358
  function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.8) {
1098
1359
  const result = [];
1360
+ // Optimasi dengan bikin map kata-kata kotor berdasarkan huruf pertama
1361
+ const profanityMap = new Map();
1362
+ // Kelompokkan kata-kata kotor berdasarkan huruf pertama biar pencariannya lebih cepat
1363
+ for (const word of profanityWords) {
1364
+ if (word.length < 1)
1365
+ continue;
1366
+ const firstChar = word[0].toLowerCase();
1367
+ if (!profanityMap.has(firstChar)) {
1368
+ profanityMap.set(firstChar, []);
1369
+ }
1370
+ profanityMap.get(firstChar).push(word);
1371
+ }
1099
1372
  const words = text.toLowerCase().split(/\s+/);
1100
1373
  for (const word of words) {
1101
1374
  if (word.length < 3)
1102
1375
  continue;
1103
- for (const profanity of profanityWords) {
1376
+ // Cuma bandingin dengan kata-kata kotor yang huruf pertamanya sama
1377
+ // atau yang perbedaan panjangnya masih masuk akal
1378
+ const firstChar = word[0];
1379
+ const candidateWords = profanityMap.get(firstChar) || [];
1380
+ // Cek juga huruf-huruf yang berdekatan (buat antisipasi typo di huruf pertama)
1381
+ // Ini opsional tapi bikin deteksinya lebih bagus
1382
+ const charCode = firstChar.charCodeAt(0);
1383
+ const prevChar = String.fromCharCode(charCode - 1);
1384
+ const nextChar = String.fromCharCode(charCode + 1);
1385
+ const adjacentCandidates = [
1386
+ ...(profanityMap.get(prevChar) || []),
1387
+ ...(profanityMap.get(nextChar) || []),
1388
+ ];
1389
+ // Gabungin kandidat-kandidatnya, prioritasin yang huruf pertamanya sama persis
1390
+ const allCandidates = [...candidateWords, ...adjacentCandidates];
1391
+ // Filter kandidat berdasarkan perbedaan panjang sebelum ngitung kemiripannya
1392
+ const lengthFilteredCandidates = allCandidates.filter((candidate) => Math.abs(candidate.length - word.length) <= 2);
1393
+ // Cari yang paling cocok
1394
+ let bestMatch = null;
1395
+ for (const profanity of lengthFilteredCandidates) {
1104
1396
  const similarity = stringSimilarity(word, profanity);
1105
- if (similarity >= threshold) {
1106
- result.push({
1397
+ if (similarity >= threshold &&
1398
+ (!bestMatch || similarity > bestMatch.similarity)) {
1399
+ bestMatch = {
1107
1400
  word,
1108
1401
  original: profanity,
1109
1402
  similarity,
1110
- });
1111
- break;
1403
+ };
1112
1404
  }
1113
1405
  }
1406
+ if (bestMatch) {
1407
+ result.push(bestMatch);
1408
+ }
1114
1409
  }
1115
1410
  return result;
1116
1411
  }
@@ -1125,30 +1420,75 @@ function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.
1125
1420
  */
1126
1421
  function findProfanityByLevenshteinDistance(text, profanityWords, threshold = 0.8, maxDistance = 2) {
1127
1422
  const result = [];
1423
+ // map kata-kata kotor dikelompokkan sesuai panjangnya
1424
+ const profanityByLength = new Map();
1425
+ // Kelompokkin kata-kata kotor berdasarkan panjangnya biar pencarian lebih cepat
1426
+ for (const word of profanityWords) {
1427
+ const length = word.length;
1428
+ if (!profanityByLength.has(length)) {
1429
+ profanityByLength.set(length, []);
1430
+ }
1431
+ profanityByLength.get(length).push(word);
1432
+ }
1128
1433
  const words = text.toLowerCase().split(/\s+/);
1129
1434
  for (const word of words) {
1130
1435
  if (word.length < 3)
1131
1436
  continue;
1132
- for (const profanity of profanityWords) {
1133
- if (Math.abs(word.length - profanity.length) > maxDistance)
1134
- continue;
1135
- const distance = levenshteinDistance(word, profanity);
1136
- if (distance <= maxDistance) {
1137
- const similarity = stringSimilarity(word, profanity);
1138
- if (similarity >= threshold) {
1139
- result.push({
1140
- word,
1141
- original: profanity,
1142
- similarity,
1143
- distance,
1144
- });
1145
- break;
1437
+ let bestMatch = null;
1438
+ for (let len = Math.max(3, word.length - maxDistance); len <= word.length + maxDistance; len++) {
1439
+ const candidates = profanityByLength.get(len) || [];
1440
+ for (const profanity of candidates) {
1441
+ if (!isCharacterCountSimilar(word, profanity, maxDistance)) {
1442
+ continue;
1443
+ }
1444
+ const distance = levenshteinDistance(word, profanity);
1445
+ if (distance <= maxDistance) {
1446
+ const similarity = 1 - distance / Math.max(word.length, profanity.length);
1447
+ if (similarity >= threshold &&
1448
+ (!bestMatch || similarity > bestMatch.similarity)) {
1449
+ bestMatch = {
1450
+ word,
1451
+ original: profanity,
1452
+ similarity,
1453
+ distance,
1454
+ };
1455
+ if (distance === 0 || similarity > 0.95) {
1456
+ break;
1457
+ }
1458
+ }
1146
1459
  }
1147
1460
  }
1148
1461
  }
1462
+ if (bestMatch) {
1463
+ result.push(bestMatch);
1464
+ }
1149
1465
  }
1150
1466
  return result;
1151
1467
  }
1468
+ /**
1469
+ * Helper function to efficiently check if character counts between two strings
1470
+ * are similar enough to warrant a full Levenshtein calculation
1471
+ */
1472
+ function isCharacterCountSimilar(str1, str2, maxDifference) {
1473
+ const charCount1 = {};
1474
+ const charCount2 = {};
1475
+ for (const char of str1) {
1476
+ charCount1[char] = (charCount1[char] || 0) + 1;
1477
+ }
1478
+ for (const char of str2) {
1479
+ charCount2[char] = (charCount2[char] || 0) + 1;
1480
+ }
1481
+ let diffCount = 0;
1482
+ for (const char in charCount1) {
1483
+ diffCount += Math.abs((charCount1[char] || 0) - (charCount2[char] || 0));
1484
+ }
1485
+ for (const char in charCount2) {
1486
+ if (!charCount1[char]) {
1487
+ diffCount += charCount2[char];
1488
+ }
1489
+ }
1490
+ return diffCount <= maxDifference * 2;
1491
+ }
1152
1492
 
1153
1493
  const DEFAULT_OPTIONS = {
1154
1494
  replaceWith: "*",
@@ -1303,6 +1643,157 @@ function makeRandomGrawlixString(length) {
1303
1643
  return result;
1304
1644
  }
1305
1645
 
1646
+ /**
1647
+ * Implementasi algoritma Aho-Corasick untuk pencocokan string
1648
+ * Algoritma ini memungkinkan pencarian banyak pola dalam satu kali pembacaan teks
1649
+ */
1650
+ class AhoCorasick {
1651
+ constructor() {
1652
+ this.built = false;
1653
+ this.root = {
1654
+ children: new Map(),
1655
+ fail: null,
1656
+ output: new Set(),
1657
+ depth: 0,
1658
+ };
1659
+ }
1660
+ /**
1661
+ * Menambahkan pola ke dalam trie
1662
+ * @param pattern Pola yang akan ditambahkan
1663
+ */
1664
+ addPattern(pattern) {
1665
+ if (this.built) {
1666
+ throw new Error("Cannot add patterns after the automaton is built");
1667
+ }
1668
+ let node = this.root;
1669
+ const normalizedPattern = pattern.toLowerCase();
1670
+ for (let i = 0; i < normalizedPattern.length; i++) {
1671
+ const char = normalizedPattern[i];
1672
+ if (!node.children.has(char)) {
1673
+ node.children.set(char, {
1674
+ children: new Map(),
1675
+ fail: null,
1676
+ output: new Set(),
1677
+ depth: node.depth + 1,
1678
+ char,
1679
+ });
1680
+ }
1681
+ node = node.children.get(char);
1682
+ }
1683
+ node.output.add(normalizedPattern);
1684
+ }
1685
+ /**
1686
+ * Membangun fungsi failure
1687
+ */
1688
+ build() {
1689
+ if (this.built)
1690
+ return;
1691
+ const queue = [];
1692
+ // Set fail pointer for depth 1 nodes to root
1693
+ for (const child of this.root.children.values()) {
1694
+ child.fail = this.root;
1695
+ queue.push(child);
1696
+ }
1697
+ // BFS to build failure links
1698
+ while (queue.length > 0) {
1699
+ const current = queue.shift();
1700
+ for (const [char, child] of current.children.entries()) {
1701
+ queue.push(child);
1702
+ let failNode = current.fail;
1703
+ // Find the longest proper suffix that is also a prefix
1704
+ while (failNode !== null && !failNode.children.has(char)) {
1705
+ failNode = failNode.fail;
1706
+ }
1707
+ if (failNode === null) {
1708
+ child.fail = this.root;
1709
+ }
1710
+ else {
1711
+ child.fail = failNode.children.get(char);
1712
+ // Add outputs from the fail state to this node
1713
+ for (const output of child.fail.output) {
1714
+ child.output.add(output);
1715
+ }
1716
+ }
1717
+ }
1718
+ }
1719
+ this.built = true;
1720
+ }
1721
+ /**
1722
+ * Mencari semua kemunculan pola dalam teks
1723
+ * @param text Teks yang akan dicari
1724
+ * @returns Map pola yang ditemukan dengan jumlah kemunculannya
1725
+ */
1726
+ search(text) {
1727
+ if (!this.built) {
1728
+ this.build();
1729
+ }
1730
+ const matches = new Map();
1731
+ const normalizedText = text.toLowerCase();
1732
+ let node = this.root;
1733
+ for (let i = 0; i < normalizedText.length; i++) {
1734
+ const char = normalizedText[i];
1735
+ // Follow failure links until we find a matching transition or reach root
1736
+ while (node !== this.root && !node.children.has(char)) {
1737
+ node = node.fail;
1738
+ }
1739
+ // Try to follow the transition
1740
+ if (node.children.has(char)) {
1741
+ node = node.children.get(char);
1742
+ }
1743
+ // Check for any matches at this node
1744
+ for (const match of node.output) {
1745
+ matches.set(match, (matches.get(match) || 0) + 1);
1746
+ }
1747
+ }
1748
+ return matches;
1749
+ }
1750
+ /**
1751
+ * Mencari semua kemunculan pola dalam teks dan mengembalikan hanya pola unik
1752
+ * @param text Teks yang akan dicari
1753
+ * @returns Set pola yang ditemukan
1754
+ */
1755
+ searchUnique(text) {
1756
+ const matches = this.search(text);
1757
+ return new Set(matches.keys());
1758
+ }
1759
+ /**
1760
+ * Mengecek apakah teks mengandung setidaknya satu pola
1761
+ * @param text Teks yang akan dicari
1762
+ * @returns Boolean apakah pola ditemukan
1763
+ */
1764
+ containsAny(text) {
1765
+ if (!this.built) {
1766
+ this.build();
1767
+ }
1768
+ const normalizedText = text.toLowerCase();
1769
+ let node = this.root;
1770
+ for (let i = 0; i < normalizedText.length; i++) {
1771
+ const char = normalizedText[i];
1772
+ while (node !== this.root && !node.children.has(char)) {
1773
+ node = node.fail;
1774
+ }
1775
+ if (node.children.has(char)) {
1776
+ node = node.children.get(char);
1777
+ }
1778
+ if (node.output.size > 0) {
1779
+ return true;
1780
+ }
1781
+ }
1782
+ return false;
1783
+ }
1784
+ }
1785
+
1786
+ const globalAhoCorasick = new AhoCorasick();
1787
+ let ahoCorasickInitialized = false;
1788
+ function initializeAhoCorasick(words) {
1789
+ if (ahoCorasickInitialized)
1790
+ return;
1791
+ for (const word of words) {
1792
+ globalAhoCorasick.addPattern(word);
1793
+ }
1794
+ globalAhoCorasick.build();
1795
+ ahoCorasickInitialized = true;
1796
+ }
1306
1797
  function findProfanity(text, options = {}) {
1307
1798
  const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, useLevenshtein = false, maxLevenshteinDistance = 2, } = { ...DEFAULT_OPTIONS, ...options };
1308
1799
  const normalizedText = normalizeText(text);
@@ -1342,24 +1833,16 @@ function findProfanity(text, options = {}) {
1342
1833
  }
1343
1834
  const matches = new Set();
1344
1835
  const actualMatches = new Map();
1345
- wordsToCheck.forEach((word) => {
1346
- const regex = createWordRegex(word, {
1347
- wholeWord: !checkSubstring,
1348
- caseSensitive: false,
1349
- leetSpeak: false,
1350
- detectSplit: false,
1351
- indonesianVariation: false,
1352
- });
1353
- let match;
1354
- while ((match = regex.exec(normalizedText)) !== null) {
1355
- const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1356
- matches.add(originalWord);
1357
- if (!actualMatches.has(originalWord)) {
1358
- actualMatches.set(originalWord, []);
1359
- }
1360
- actualMatches.get(originalWord)?.push(match[0]);
1836
+ initializeAhoCorasick(wordsToCheck);
1837
+ const basicMatches = globalAhoCorasick.searchUnique(normalizedText);
1838
+ for (const match of basicMatches) {
1839
+ const originalWord = aliasMap.get(match.toLowerCase()) || match.toLowerCase();
1840
+ matches.add(originalWord);
1841
+ if (!actualMatches.has(originalWord)) {
1842
+ actualMatches.set(originalWord, []);
1361
1843
  }
1362
- });
1844
+ actualMatches.get(originalWord)?.push(match);
1845
+ }
1363
1846
  if (detectLeetSpeak) {
1364
1847
  wordsToCheck.forEach((word) => {
1365
1848
  const leetRegex = createWordRegex(word, {
@@ -1536,7 +2019,7 @@ function calculateSeverity(matchDetails) {
1536
2019
  * @returns FilterResult dengan hasil filter
1537
2020
  */
1538
2021
  function filter(text, options = {}) {
1539
- const { replaceWith = '*', fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, detectSplit = false, detectSimilarity = false, useLevenshtein = false, maxLevenshteinDistance = 2, similarityThreshold = 0.8, } = { ...DEFAULT_OPTIONS, ...options };
2022
+ const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, detectSplit = false, detectSimilarity = false, useLevenshtein = false, maxLevenshteinDistance = 2, similarityThreshold = 0.8, } = { ...DEFAULT_OPTIONS, ...options };
1540
2023
  const matches = findProfanity(text, {
1541
2024
  ...options,
1542
2025
  detectLeetSpeak,
@@ -1568,7 +2051,7 @@ function filter(text, options = {}) {
1568
2051
  variants.push(word);
1569
2052
  const uniqueVariants = [...new Set(variants)];
1570
2053
  uniqueVariants.forEach((variant) => {
1571
- const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, 'gi');
2054
+ const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
1572
2055
  let match;
1573
2056
  while ((match = regex.exec(filteredText)) !== null) {
1574
2057
  const originalWord = match[0];
@@ -1586,7 +2069,7 @@ function filter(text, options = {}) {
1586
2069
  censored: censoredWord,
1587
2070
  metadata,
1588
2071
  });
1589
- filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'), censoredWord);
2072
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"), censoredWord);
1590
2073
  }
1591
2074
  });
1592
2075
  if (detectSplit || detectLeetSpeak) {
@@ -1615,7 +2098,7 @@ function filter(text, options = {}) {
1615
2098
  censored: censoredWord,
1616
2099
  metadata,
1617
2100
  });
1618
- filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), 'g'), censoredWord);
2101
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), "g"), censoredWord);
1619
2102
  }
1620
2103
  }
1621
2104
  if (detectSplit) {
@@ -1643,7 +2126,7 @@ function filter(text, options = {}) {
1643
2126
  censored: censoredWord,
1644
2127
  metadata,
1645
2128
  });
1646
- filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), 'g'), censoredWord);
2129
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), "g"), censoredWord);
1647
2130
  }
1648
2131
  }
1649
2132
  }
@@ -1655,7 +2138,7 @@ function filter(text, options = {}) {
1655
2138
  m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
1656
2139
  const variants = actualMatches.get(word.toLowerCase()) || [];
1657
2140
  variants.forEach((variant) => {
1658
- const exactVariantRegex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, 'gi');
2141
+ const exactVariantRegex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
1659
2142
  let match;
1660
2143
  while ((match = exactVariantRegex.exec(filteredText)) !== null) {
1661
2144
  const originalWord = match[0];
@@ -1673,7 +2156,7 @@ function filter(text, options = {}) {
1673
2156
  censored: censoredWord,
1674
2157
  metadata,
1675
2158
  });
1676
- filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'), censoredWord);
2159
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"), censoredWord);
1677
2160
  }
1678
2161
  });
1679
2162
  });
@@ -1722,8 +2205,20 @@ function analyze(text, options = {}) {
1722
2205
  const severityScore = calculateSeverity(matchDetails);
1723
2206
  let similarWords = [];
1724
2207
  if (mergedOptions.detectSimilarity) {
1725
- const wordList = matchDetails.map((word) => word.word);
1726
- similarWords = findPossibleProfanityBySimiliarity(text, wordList, mergedOptions.similarityThreshold || 0.8);
2208
+ if (matchDetails.length > 0) {
2209
+ const wordList = matchDetails.map((word) => word.word);
2210
+ if (mergedOptions.useLevenshtein) {
2211
+ const levenshteinResults = findProfanityByLevenshteinDistance(text, wordList, mergedOptions.similarityThreshold || 0.8, mergedOptions.maxLevenshteinDistance || 2);
2212
+ similarWords = levenshteinResults.map((item) => ({
2213
+ word: item.word,
2214
+ original: item.original,
2215
+ similarity: item.similarity,
2216
+ }));
2217
+ }
2218
+ else {
2219
+ similarWords = findPossibleProfanityBySimiliarity(text, wordList, mergedOptions.similarityThreshold || 0.8);
2220
+ }
2221
+ }
1727
2222
  }
1728
2223
  return {
1729
2224
  hasProfanity: true,
@@ -1820,9 +2315,9 @@ function analyzeWithContext(text, contextWindowSize = 5, options = {}) {
1820
2315
  const regex = createContextRegex(word, contextWindowSize);
1821
2316
  let match;
1822
2317
  while ((match = regex.exec(text)) !== null) {
1823
- const beforeContext = match[1] || '';
2318
+ const beforeContext = match[1] || "";
1824
2319
  const wordMatch = match[2];
1825
- const afterContext = match[3] || '';
2320
+ const afterContext = match[3] || "";
1826
2321
  result.push({
1827
2322
  word: wordMatch,
1828
2323
  context: beforeContext + wordMatch + afterContext,