@sideid/id-profanity-filter 1.9.5 → 1.10.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/.eslintrc.js +44 -16
  2. package/.github/workflows/release.yml +62 -0
  3. package/CONTRIBUTING.md +150 -150
  4. package/LICENSE +21 -21
  5. package/README.md +548 -285
  6. package/dist/config/options.d.ts +24 -0
  7. package/dist/constants/categories/blasphemy.d.ts +4 -0
  8. package/dist/constants/categories/disgusting.d.ts +4 -0
  9. package/dist/constants/categories/drugs.d.ts +4 -0
  10. package/dist/constants/categories/profanity.d.ts +4 -0
  11. package/dist/constants/categories/slur.d.ts +4 -0
  12. package/dist/index.d.ts +2 -0
  13. package/dist/index.esm.js +923 -94
  14. package/dist/index.esm.js.map +1 -1
  15. package/dist/index.js +924 -93
  16. package/dist/index.js.map +1 -1
  17. package/dist/types/index.d.ts +2 -0
  18. package/dist/utils/ahoCorasick.d.ts +36 -0
  19. package/dist/utils/similarityUtils.d.ts +35 -0
  20. package/eslint.config.mjs +40 -0
  21. package/examples/advanced.ts +120 -0
  22. package/examples/basic.ts +71 -52
  23. package/examples/custom-list.ts +140 -0
  24. package/jest.config.mjs +10 -10
  25. package/package.json +3 -2
  26. package/prettierrc +6 -6
  27. package/rollup.config.mjs +35 -35
  28. package/src/config/options.ts +2 -0
  29. package/src/constants/categories/blasphemy.ts +25 -0
  30. package/src/constants/categories/disgusting.ts +82 -0
  31. package/src/constants/categories/drugs.ts +72 -0
  32. package/src/constants/categories/profanity.ts +139 -0
  33. package/src/constants/categories/slur.ts +102 -0
  34. package/src/constants/regions/general.ts +111 -2
  35. package/src/constants/regions/jawa.ts +257 -3
  36. package/src/constants/wordList.ts +15 -8
  37. package/src/core/analyzer.ts +28 -13
  38. package/src/core/filter.ts +178 -37
  39. package/src/core/matcher.ts +146 -69
  40. package/src/index.ts +21 -2
  41. package/src/types/index.ts +4 -2
  42. package/src/utils/ahoCorasick.ts +179 -0
  43. package/src/utils/regexUtils.ts +0 -1
  44. package/src/utils/similarityUtils.ts +239 -7
  45. package/tsconfig.json +115 -115
  46. package/.github/workflows/ci.yml +0 -0
  47. package/src/constants/categories/index.ts +0 -31
  48. package/src/constants/regions/index.ts +0 -62
package/dist/index.esm.js CHANGED
@@ -4,7 +4,7 @@ const general = [
4
4
  category: "profanity",
5
5
  region: "general",
6
6
  severity: 0.7,
7
- aliases: ["anjay", "anjir", "anying", "njing", "anj"],
7
+ aliases: ["anjay", "anjir", "anying", "njing", "anj", "anjg", "ajg"],
8
8
  description: "Mengacu pada hewan anjing, digunakan sebagai umpatan",
9
9
  context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
10
10
  },
@@ -13,7 +13,7 @@ const general = [
13
13
  category: "profanity",
14
14
  region: "general",
15
15
  severity: 0.6,
16
- aliases: ["bab1", "b4b1"],
16
+ aliases: ["bab1", "b4b1", "b4bi", "8481", "8ab1", "ba81"],
17
17
  description: "Mengacu pada hewan babi, digunakan sebagai umpatan",
18
18
  context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
19
19
  },
@@ -98,6 +98,114 @@ const general = [
98
98
  description: "Kata yang mengacu pada orang yang banyak bicara",
99
99
  context: "Hinaan untuk menyebut orang yang banyak bicara atau cerewet",
100
100
  },
101
+ {
102
+ word: "ngentot",
103
+ category: "sexual",
104
+ region: "general",
105
+ severity: 0.9,
106
+ aliases: ["ngentod", "ntot", "tod"],
107
+ description: "Istilah kasar untuk aktivitas seksual",
108
+ context: "Kata vulgar yang merujuk pada aktivitas seksual",
109
+ },
110
+ {
111
+ word: "sialan",
112
+ category: "insult",
113
+ region: "general",
114
+ severity: 0.5,
115
+ aliases: ["sialn", "sl"],
116
+ description: "Kata yang mengacu pada orang yang membawa sial",
117
+ context: "Hinaan untuk menyebut orang yang dianggap membawa sial",
118
+ },
119
+ {
120
+ word: "pler",
121
+ category: "sexual",
122
+ region: "general",
123
+ severity: 0.9,
124
+ aliases: ["peler", "plr", "biji"],
125
+ description: "Istilah kasar untuk alat kelamin laki-laki",
126
+ context: "Kata vulgar yang merujuk pada alat kelamin laki-laki",
127
+ },
128
+ {
129
+ word: "bokep",
130
+ category: "sexual",
131
+ region: "general",
132
+ severity: 0.7,
133
+ aliases: ["bkp", "bokap"],
134
+ description: "Istilah untuk video atau konten pornografi",
135
+ context: "Kata yang mengacu pada materi pornografi",
136
+ },
137
+ {
138
+ word: "coli",
139
+ category: "sexual",
140
+ region: "general",
141
+ severity: 0.8,
142
+ aliases: ["col", "coly"],
143
+ description: "Istilah untuk masturbasi laki-laki",
144
+ context: "Kata vulgar yang merujuk pada aktivitas seksual pribadi",
145
+ },
146
+ {
147
+ word: "desah",
148
+ category: "sexual",
149
+ region: "general",
150
+ severity: 0.6,
151
+ aliases: ["ds4h", "dsh"],
152
+ description: "Istilah untuk suara yang dibuat selama aktivitas seksual",
153
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
154
+ },
155
+ {
156
+ word: "seks",
157
+ category: "sexual",
158
+ region: "general",
159
+ severity: 0.5,
160
+ aliases: ["sex", "ML"],
161
+ description: "Istilah untuk aktivitas seksual",
162
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
163
+ },
164
+ {
165
+ word: "kondom",
166
+ category: "sexual",
167
+ region: "general",
168
+ severity: 0.5,
169
+ aliases: ["kndm", "kondom", "cd"],
170
+ description: "Alat kontrasepsi",
171
+ context: "Dapat menjadi vulgar tergantung konteks penggunaan",
172
+ },
173
+ {
174
+ word: "ngewe",
175
+ category: "sexual",
176
+ region: "general",
177
+ severity: 0.9,
178
+ aliases: ["ngew", "we"],
179
+ description: "Istilah kasar untuk aktivitas seksual",
180
+ context: "Kata vulgar yang merujuk pada aktivitas seksual",
181
+ },
182
+ {
183
+ word: "puki",
184
+ category: "sexual",
185
+ region: "general",
186
+ severity: 0.9,
187
+ aliases: ["puk", "pukih"],
188
+ description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
189
+ context: "Kata vulgar yang merujuk pada anatomi seksual",
190
+ },
191
+ {
192
+ word: "xxx",
193
+ category: "sexual",
194
+ region: "general",
195
+ severity: 0.6,
196
+ aliases: ["xXx", "triplex"],
197
+ description: "Simbol yang sering digunakan untuk menandai konten pornografi",
198
+ context: "Digunakan untuk menandai konten seksual eksplisit",
199
+ },
200
+ {
201
+ word: "xnxx",
202
+ category: "sexual",
203
+ region: "general",
204
+ severity: 0.6,
205
+ aliases: ["xnxx", "xnx"],
206
+ description: "simbol yang sering digunakan untuk menandai konten pornografi",
207
+ context: "Digunakan untuk menandai konten seksual eksplisit",
208
+ }
101
209
  ];
102
210
  general.map((item) => item.word);
103
211
 
@@ -115,7 +223,7 @@ const jawa = [
115
223
  word: "jancok",
116
224
  category: "sexual",
117
225
  region: "jawa",
118
- severity: 0.8,
226
+ severity: 0.9,
119
227
  aliases: ["jancuk", "jncok", "jancuk", "jncuk", "dancok", "dancuk"],
120
228
  description: "Kata umpatan kasar dalam Bahasa Jawa",
121
229
  context: "Umpatan kasar yang umum digunakan di Jawa Timur",
@@ -151,7 +259,7 @@ const jawa = [
151
259
  word: "mbokne ancok",
152
260
  category: "insult",
153
261
  region: "jawa",
154
- severity: 0.8,
262
+ severity: 0.9,
155
263
  aliases: ["mbokne", "mbokneancok"],
156
264
  description: "Umpatan yang menyinggung ibu seseorang",
157
265
  context: "Umpatan kasar yang menyinggung orangtua orang lain",
@@ -160,7 +268,7 @@ const jawa = [
160
268
  word: "pekok",
161
269
  category: "insult",
162
270
  region: "jawa",
163
- severity: 0.6,
271
+ severity: 0.7,
164
272
  aliases: ["pekak", "pekilk"],
165
273
  description: "Kata hinaan yang menunjukkan kebodohan",
166
274
  context: "Hinaan untuk menyebut orang yang dianggap sangat bodoh",
@@ -192,6 +300,257 @@ const jawa = [
192
300
  description: "Secara harfiah berarti kepalamu, digunakan sebagai umpatan ringan",
193
301
  context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
194
302
  },
303
+ {
304
+ word: "kontol",
305
+ category: "sexual",
306
+ region: "jawa",
307
+ severity: 0.8,
308
+ aliases: ["kntl", "kontl"],
309
+ description: "Mengacu ke alat kelamin laki-laki",
310
+ context: "Kata vulgar yang sering digunakan sebagai umpatan kasar",
311
+ },
312
+ {
313
+ word: "tempek",
314
+ category: "sexual",
315
+ region: "jawa",
316
+ severity: 0.8,
317
+ aliases: ["mpek", "torok", "tempk"],
318
+ description: "Mengacu pada alat kelamin perempuan",
319
+ context: "Kata vulgar yang digunakan sebagai umpatan atau hinaan",
320
+ },
321
+ {
322
+ word: "silit",
323
+ category: "insult",
324
+ region: "jawa",
325
+ severity: 0.6,
326
+ aliases: ["selet", "tilis"],
327
+ description: "Mengacu pada bagian dubur atau anus",
328
+ context: "Kata kasar yang digunakan sebagai hinaan",
329
+ },
330
+ {
331
+ word: "mbahmu",
332
+ category: "insult",
333
+ region: "jawa",
334
+ severity: 0.6,
335
+ aliases: ["mbahmu kiper", "mbah mu"],
336
+ description: "Hinaan yang menyinggung nenek/kakek seseorang",
337
+ context: "Umpatan ringan untuk menanggapi sesuatu yang tidak masuk akal",
338
+ },
339
+ {
340
+ word: "makmu",
341
+ category: "insult",
342
+ region: "jawa",
343
+ severity: 0.7,
344
+ aliases: ["mak mu", "mamamu"],
345
+ description: "Hinaan yang menyinggung ibu seseorang",
346
+ context: "Umpatan yang dianggap kasar karena menyinggung orang tua",
347
+ },
348
+ {
349
+ word: "bajingan",
350
+ category: "insult",
351
+ region: "jawa",
352
+ severity: 0.7,
353
+ aliases: [
354
+ "bajilak",
355
+ "bajhingan",
356
+ "bajingak",
357
+ "bajingseng",
358
+ "bajindul",
359
+ "bajigur",
360
+ "jingan",
361
+ ],
362
+ description: "Sebutan untuk orang yang dianggap jahat atau tidak bermoral",
363
+ context: "Umpatan untuk mengekspresikan kemarahan atau kekesalan",
364
+ },
365
+ {
366
+ word: "cocote",
367
+ category: "insult",
368
+ region: "jawa",
369
+ severity: 0.6,
370
+ aliases: ["cocot", "bacot", "nyocot"],
371
+ description: "Mengacu pada mulut dengan konotasi negatif",
372
+ context: "Umpatan untuk menyuruh seseorang berhenti berbicara",
373
+ },
374
+ {
375
+ word: "ngentot",
376
+ category: "sexual",
377
+ region: "jawa",
378
+ severity: 0.9,
379
+ aliases: ["kentu", "kentot", "iclik", "ngtt", "iclk"],
380
+ description: "Mengacu pada aktivitas seksual",
381
+ context: "Kata vulgar yang digunakan sebagai umpatan kasar",
382
+ },
383
+ {
384
+ word: "edan",
385
+ category: "insult",
386
+ region: "jawa",
387
+ severity: 0.5,
388
+ aliases: ["gendeng", "gila", "gendheng", "sarap"],
389
+ description: "Secara harfiah berarti gila atau tidak waras",
390
+ context: "Umpatan untuk menyebut seseorang yang dianggap tidak masuk akal",
391
+ },
392
+ {
393
+ word: "dapuranmu",
394
+ category: "insult",
395
+ region: "jawa",
396
+ severity: 0.6,
397
+ aliases: ["raimu", "rai mu"],
398
+ description: "Secara harfiah mengacu pada wajah atau rupa seseorang",
399
+ context: "Umpatan untuk menghina penampilan atau wajah seseorang",
400
+ },
401
+ {
402
+ word: "damput",
403
+ category: "insult",
404
+ region: "jawa",
405
+ severity: 0.7,
406
+ aliases: ["diamput"],
407
+ description: "Variasi bentuk umpatan dengan makna serupa dengan jancok",
408
+ context: "Umpatan kasar untuk mengekspresikan kemarahan",
409
+ },
410
+ {
411
+ word: "mbathang",
412
+ category: "insult",
413
+ region: "jawa",
414
+ severity: 0.7,
415
+ aliases: ["mbatang"],
416
+ description: "Secara harfiah berarti bangkai",
417
+ context: "Umpatan kasar untuk menghina seseorang",
418
+ },
419
+ {
420
+ word: "ndlogok",
421
+ category: "insult",
422
+ region: "jawa",
423
+ severity: 0.6,
424
+ aliases: ["ndelodok", "ndlodok"],
425
+ description: "Mengacu pada tindakan yang dianggap bodoh atau tidak masuk akal",
426
+ context: "Hinaan untuk mengkritik tindakan seseorang",
427
+ },
428
+ {
429
+ word: "nggateli",
430
+ category: "insult",
431
+ region: "jawa",
432
+ severity: 0.5,
433
+ aliases: ["gateli", "gathel"],
434
+ description: "Secara harfiah berarti gatal atau menyebalkan",
435
+ context: "Ungkapan untuk menunjukkan kekesalan terhadap perilaku seseorang",
436
+ },
437
+ {
438
+ word: "perek",
439
+ category: "sexual",
440
+ region: "jawa",
441
+ severity: 0.8,
442
+ aliases: ["lonthe", "pelacur"],
443
+ description: "Istilah merendahkan untuk pekerja seks komersial",
444
+ context: "Kata kasar untuk menghina wanita",
445
+ },
446
+ {
447
+ word: "picek",
448
+ category: "insult",
449
+ region: "jawa",
450
+ severity: 0.6,
451
+ aliases: ["pcek", "buta"],
452
+ description: "Secara harfiah berarti buta atau tidak bisa melihat",
453
+ context: "Hinaan untuk orang yang dianggap tidak bisa melihat kenyataan",
454
+ },
455
+ {
456
+ word: "untumu",
457
+ category: "insult",
458
+ region: "jawa",
459
+ severity: 0.5,
460
+ aliases: ["gigimu", "untu mu"],
461
+ description: "Secara harfiah berarti gigimu",
462
+ context: "Umpatan ringan untuk menanggapi sesuatu yang tidak disetujui",
463
+ },
464
+ {
465
+ word: "goblog",
466
+ category: "insult",
467
+ region: "jawa",
468
+ severity: 0.7,
469
+ aliases: ["ghoblog", "goblok", "gobhlok", "pekok"],
470
+ description: "Kata hinaan yang menunjukkan kebodohan ekstrem",
471
+ context: "Hinaan untuk menyebut seseorang yang dianggap sangat bodoh",
472
+ },
473
+ {
474
+ word: "tolol",
475
+ category: "insult",
476
+ region: "jawa",
477
+ severity: 0.7,
478
+ aliases: ["tholol", "tlol"],
479
+ description: "Kata hinaan yang menunjukkan kebodohan",
480
+ context: "Hinaan untuk menyebut seseorang yang dianggap bodoh",
481
+ },
482
+ {
483
+ word: "budheg",
484
+ category: "insult",
485
+ region: "jawa",
486
+ severity: 0.6,
487
+ aliases: ["budeg", "bdeg"],
488
+ description: "Secara harfiah berarti tuli atau tidak bisa mendengar",
489
+ context: "Hinaan untuk orang yang dianggap tidak mau mendengarkan",
490
+ },
491
+ {
492
+ word: "jiangkrik",
493
+ category: "insult",
494
+ region: "jawa",
495
+ severity: 0.4,
496
+ aliases: ["jiangkrek", "jangkrik"],
497
+ description: "Secara harfiah berarti jangkrik, digunakan sebagai eufemisme",
498
+ context: "Umpatan ringan sebagai pengganti kata kasar yang lebih vulgar",
499
+ },
500
+ {
501
+ word: "diamput",
502
+ category: "insult",
503
+ region: "jawa",
504
+ severity: 0.8,
505
+ aliases: ["damput", "djamput"],
506
+ description: "Bentuk umpatan kasar dengan makna serupa jancok",
507
+ context: "Kata kasar untuk mengekspresikan kemarahan",
508
+ },
509
+ {
510
+ word: "celeng",
511
+ category: "insult",
512
+ region: "jawa",
513
+ severity: 0.6,
514
+ aliases: ["cleng", "babi hutan"],
515
+ description: "Secara harfiah berarti babi hutan",
516
+ context: "Hinaan untuk orang yang dianggap jorok atau rakus",
517
+ },
518
+ {
519
+ word: "kampret",
520
+ category: "insult",
521
+ region: "jawa",
522
+ severity: 0.5,
523
+ aliases: ["kmpret", "kmprt"],
524
+ description: "Secara harfiah berarti kelelawar kecil",
525
+ context: "Umpatan ringan untuk mengekspresikan kekesalan",
526
+ },
527
+ {
528
+ word: "ndeso",
529
+ category: "insult",
530
+ region: "jawa",
531
+ severity: 0.4,
532
+ aliases: ["ndesa", "deso"],
533
+ description: "Secara harfiah berarti dari desa atau kampungan",
534
+ context: "Hinaan untuk orang yang dianggap kurang modern atau berpendidikan",
535
+ },
536
+ {
537
+ word: "kere",
538
+ category: "insult",
539
+ region: "jawa",
540
+ severity: 0.5,
541
+ aliases: ["miskin", "mlarat"],
542
+ description: "Secara harfiah berarti miskin atau tidak punya uang",
543
+ context: "Hinaan untuk status ekonomi seseorang yang dianggap rendah",
544
+ },
545
+ {
546
+ word: "itil",
547
+ category: "sexual",
548
+ region: "jawa",
549
+ severity: 0.9,
550
+ aliases: ["itl", "itul"],
551
+ description: "Kata vulgar yang mengacu pada bagian dari alat kelamin perempuan",
552
+ context: "Kata vulgar yang merujuk pada anatomi seksual",
553
+ },
195
554
  ];
196
555
  jawa.map((item) => item.word);
197
556
 
@@ -712,7 +1071,6 @@ function addLeetSpeakVariations(pattern) {
712
1071
  t: ["t", "7", "+"],
713
1072
  z: ["z", "2"],
714
1073
  };
715
- // Ganti tiap karakter dengan variasinya dalam grup character class
716
1074
  return pattern
717
1075
  .split("")
718
1076
  .map((char) => {
@@ -896,6 +1254,37 @@ function findMostSimilar(target, candidates, threshold = 0.7) {
896
1254
  }
897
1255
  return mostSimilar;
898
1256
  }
1257
+ /**
1258
+ * Mencari string yang paling mirip dari array menggunakan Levenshtein distance
1259
+ *
1260
+ * @param target String target
1261
+ * @param candidates Array string kandidat
1262
+ * @param threshold Minimum kesamaan yang diterima (0-1)
1263
+ * @param maxDistance Jarak Levenshtein maksimal yang diterima (default: 3)
1264
+ * @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
1265
+ */
1266
+ function findMostSimilarWithLevenshtein(target, candidates, threshold = 0.7, maxDistance = 3) {
1267
+ if (!candidates.length)
1268
+ return null;
1269
+ let maxSimilarity = 0;
1270
+ let minDistance = Infinity;
1271
+ let mostSimilar = null;
1272
+ for (const candidate of candidates) {
1273
+ if (Math.abs(target.length - candidate.length) > maxDistance)
1274
+ continue;
1275
+ const distance = levenshteinDistance(target, candidate);
1276
+ const similarity = stringSimilarity(target, candidate);
1277
+ if ((similarity > maxSimilarity && similarity >= threshold) ||
1278
+ (similarity >= threshold && distance < minDistance)) {
1279
+ maxSimilarity = similarity;
1280
+ minDistance = distance;
1281
+ mostSimilar = candidate;
1282
+ if (distance <= 1 || similarity > 0.95)
1283
+ break;
1284
+ }
1285
+ }
1286
+ return mostSimilar;
1287
+ }
899
1288
  /**
900
1289
  * Cek apakah string mungkin merupakan variasi dari kata kotor
901
1290
  * menggunakan kesamaan string
@@ -946,34 +1335,156 @@ function clusterSimilarWords(words, threshold = 0.8) {
946
1335
  }
947
1336
  /**
948
1337
  * Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
1338
+ * dengan optimasi untuk mengurangi kompleksitas
949
1339
  *
950
1340
  * @param text Teks yang akan diperiksa
951
1341
  * @param profanityWords Daftar kata kotor
952
1342
  * @param threshold Batas minimum kesamaan (default: 0.8)
953
1343
  * @returns Array kata yang mungkin merupakan kata kotor
954
1344
  */
1345
+ /**
1346
+ * Cari kata-kata kotor yang mungkin dari teks berdasarkan kemiripan string
1347
+ * dengan optimasi biar prosesnya nggak terlalu berat
1348
+ *
1349
+ * @param text Teks yang mau dicek
1350
+ * @param profanityWords Daftar kata-kata kotor/kasar
1351
+ * @param threshold Batas minimal kemiripan (default: 0.8)
1352
+ * @returns Array kata yang kemungkinan kata kotor/kasar
1353
+ */
955
1354
  function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.8) {
956
1355
  const result = [];
957
- // Pisahkan teks menjadi kata-kata
1356
+ // Optimasi dengan bikin map kata-kata kotor berdasarkan huruf pertama
1357
+ const profanityMap = new Map();
1358
+ // Kelompokkan kata-kata kotor berdasarkan huruf pertama biar pencariannya lebih cepat
1359
+ for (const word of profanityWords) {
1360
+ if (word.length < 1)
1361
+ continue;
1362
+ const firstChar = word[0].toLowerCase();
1363
+ if (!profanityMap.has(firstChar)) {
1364
+ profanityMap.set(firstChar, []);
1365
+ }
1366
+ profanityMap.get(firstChar).push(word);
1367
+ }
958
1368
  const words = text.toLowerCase().split(/\s+/);
959
1369
  for (const word of words) {
960
- // Lewati kata-kata yang terlalu pendek
961
1370
  if (word.length < 3)
962
1371
  continue;
963
- for (const profanity of profanityWords) {
1372
+ // Cuma bandingin dengan kata-kata kotor yang huruf pertamanya sama
1373
+ // atau yang perbedaan panjangnya masih masuk akal
1374
+ const firstChar = word[0];
1375
+ const candidateWords = profanityMap.get(firstChar) || [];
1376
+ // Cek juga huruf-huruf yang berdekatan (buat antisipasi typo di huruf pertama)
1377
+ // Ini opsional tapi bikin deteksinya lebih bagus
1378
+ const charCode = firstChar.charCodeAt(0);
1379
+ const prevChar = String.fromCharCode(charCode - 1);
1380
+ const nextChar = String.fromCharCode(charCode + 1);
1381
+ const adjacentCandidates = [
1382
+ ...(profanityMap.get(prevChar) || []),
1383
+ ...(profanityMap.get(nextChar) || []),
1384
+ ];
1385
+ // Gabungin kandidat-kandidatnya, prioritasin yang huruf pertamanya sama persis
1386
+ const allCandidates = [...candidateWords, ...adjacentCandidates];
1387
+ // Filter kandidat berdasarkan perbedaan panjang sebelum ngitung kemiripannya
1388
+ const lengthFilteredCandidates = allCandidates.filter((candidate) => Math.abs(candidate.length - word.length) <= 2);
1389
+ // Cari yang paling cocok
1390
+ let bestMatch = null;
1391
+ for (const profanity of lengthFilteredCandidates) {
964
1392
  const similarity = stringSimilarity(word, profanity);
965
- if (similarity >= threshold) {
966
- result.push({
1393
+ if (similarity >= threshold &&
1394
+ (!bestMatch || similarity > bestMatch.similarity)) {
1395
+ bestMatch = {
967
1396
  word,
968
1397
  original: profanity,
969
1398
  similarity,
970
- });
971
- break;
1399
+ };
972
1400
  }
973
1401
  }
1402
+ if (bestMatch) {
1403
+ result.push(bestMatch);
1404
+ }
974
1405
  }
975
1406
  return result;
976
1407
  }
1408
+ /**
1409
+ * Cari kata-kata kotor yang mungkin dari teks menggunakan Levenshtein distance
1410
+ *
1411
+ * @param text Teks yang akan diperiksa
1412
+ * @param profanityWords Daftar kata kotor
1413
+ * @param threshold Batas minimum kesamaan (default: 0.8)
1414
+ * @param maxDistance Jarak Levenshtein maksimal (default: 2)
1415
+ * @returns Array kata yang mungkin merupakan kata kotor
1416
+ */
1417
+ function findProfanityByLevenshteinDistance(text, profanityWords, threshold = 0.8, maxDistance = 2) {
1418
+ const result = [];
1419
+ // map kata-kata kotor dikelompokkan sesuai panjangnya
1420
+ const profanityByLength = new Map();
1421
+ // Kelompokkin kata-kata kotor berdasarkan panjangnya biar pencarian lebih cepat
1422
+ for (const word of profanityWords) {
1423
+ const length = word.length;
1424
+ if (!profanityByLength.has(length)) {
1425
+ profanityByLength.set(length, []);
1426
+ }
1427
+ profanityByLength.get(length).push(word);
1428
+ }
1429
+ const words = text.toLowerCase().split(/\s+/);
1430
+ for (const word of words) {
1431
+ if (word.length < 3)
1432
+ continue;
1433
+ let bestMatch = null;
1434
+ for (let len = Math.max(3, word.length - maxDistance); len <= word.length + maxDistance; len++) {
1435
+ const candidates = profanityByLength.get(len) || [];
1436
+ for (const profanity of candidates) {
1437
+ if (!isCharacterCountSimilar(word, profanity, maxDistance)) {
1438
+ continue;
1439
+ }
1440
+ const distance = levenshteinDistance(word, profanity);
1441
+ if (distance <= maxDistance) {
1442
+ const similarity = 1 - distance / Math.max(word.length, profanity.length);
1443
+ if (similarity >= threshold &&
1444
+ (!bestMatch || similarity > bestMatch.similarity)) {
1445
+ bestMatch = {
1446
+ word,
1447
+ original: profanity,
1448
+ similarity,
1449
+ distance,
1450
+ };
1451
+ if (distance === 0 || similarity > 0.95) {
1452
+ break;
1453
+ }
1454
+ }
1455
+ }
1456
+ }
1457
+ }
1458
+ if (bestMatch) {
1459
+ result.push(bestMatch);
1460
+ }
1461
+ }
1462
+ return result;
1463
+ }
1464
+ /**
1465
+ * Helper function to efficiently check if character counts between two strings
1466
+ * are similar enough to warrant a full Levenshtein calculation
1467
+ */
1468
+ function isCharacterCountSimilar(str1, str2, maxDifference) {
1469
+ const charCount1 = {};
1470
+ const charCount2 = {};
1471
+ for (const char of str1) {
1472
+ charCount1[char] = (charCount1[char] || 0) + 1;
1473
+ }
1474
+ for (const char of str2) {
1475
+ charCount2[char] = (charCount2[char] || 0) + 1;
1476
+ }
1477
+ let diffCount = 0;
1478
+ for (const char in charCount1) {
1479
+ diffCount += Math.abs((charCount1[char] || 0) - (charCount2[char] || 0));
1480
+ }
1481
+ for (const char in charCount2) {
1482
+ if (!charCount1[char]) {
1483
+ diffCount += charCount2[char];
1484
+ }
1485
+ }
1486
+ return diffCount <= maxDifference * 2;
1487
+ }
977
1488
 
978
1489
  const DEFAULT_OPTIONS = {
979
1490
  replaceWith: "*",
@@ -982,6 +1493,8 @@ const DEFAULT_OPTIONS = {
982
1493
  checkSubstring: false,
983
1494
  whitelist: [],
984
1495
  severityThreshold: 0,
1496
+ useLevenshtein: false,
1497
+ maxLevenshteinDistance: 2,
985
1498
  };
986
1499
  const FILTER_PRESETS = {
987
1500
  strict: {
@@ -1126,44 +1639,206 @@ function makeRandomGrawlixString(length) {
1126
1639
  return result;
1127
1640
  }
1128
1641
 
1129
- function findProfanity(text, options = {}) {
1130
- const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, } = { ...DEFAULT_OPTIONS, ...options };
1131
- const normalizedText = normalizeText(text);
1132
- let wordsToCheck = wordList.length > 0 ? wordList : [];
1133
- if (wordsToCheck.length === 0) {
1134
- if (categories || regions || severityThreshold > 0) {
1135
- wordsToCheck = wordObjects
1136
- .filter((word) => {
1137
- const matchCategory = categories
1138
- ? categories.includes(word.category)
1139
- : true;
1140
- const matchRegion = regions ? regions.includes(word.region) : true;
1141
- const matchSeverity = word.severity >= severityThreshold;
1142
- return matchCategory && matchRegion && matchSeverity;
1143
- })
1144
- .map((word) => word.word);
1642
+ /**
1643
+ * Implementasi algoritma Aho-Corasick untuk pencocokan string
1644
+ * Algoritma ini memungkinkan pencarian banyak pola dalam satu kali pembacaan teks
1645
+ */
1646
+ class AhoCorasick {
1647
+ constructor() {
1648
+ this.built = false;
1649
+ this.root = {
1650
+ children: new Map(),
1651
+ fail: null,
1652
+ output: new Set(),
1653
+ depth: 0,
1654
+ };
1655
+ }
1656
+ /**
1657
+ * Menambahkan pola ke dalam trie
1658
+ * @param pattern Pola yang akan ditambahkan
1659
+ */
1660
+ addPattern(pattern) {
1661
+ if (this.built) {
1662
+ throw new Error("Cannot add patterns after the automaton is built");
1145
1663
  }
1146
- else {
1147
- wordsToCheck = wordObjects.map((word) => word.word);
1664
+ let node = this.root;
1665
+ const normalizedPattern = pattern.toLowerCase();
1666
+ for (let i = 0; i < normalizedPattern.length; i++) {
1667
+ const char = normalizedPattern[i];
1668
+ if (!node.children.has(char)) {
1669
+ node.children.set(char, {
1670
+ children: new Map(),
1671
+ fail: null,
1672
+ output: new Set(),
1673
+ depth: node.depth + 1,
1674
+ char,
1675
+ });
1676
+ }
1677
+ node = node.children.get(char);
1678
+ }
1679
+ node.output.add(normalizedPattern);
1680
+ }
1681
+ /**
1682
+ * Membangun fungsi failure
1683
+ */
1684
+ build() {
1685
+ if (this.built)
1686
+ return;
1687
+ const queue = [];
1688
+ // Set fail pointer for depth 1 nodes to root
1689
+ for (const child of this.root.children.values()) {
1690
+ child.fail = this.root;
1691
+ queue.push(child);
1692
+ }
1693
+ // BFS to build failure links
1694
+ while (queue.length > 0) {
1695
+ const current = queue.shift();
1696
+ for (const [char, child] of current.children.entries()) {
1697
+ queue.push(child);
1698
+ let failNode = current.fail;
1699
+ // Find the longest proper suffix that is also a prefix
1700
+ while (failNode !== null && !failNode.children.has(char)) {
1701
+ failNode = failNode.fail;
1702
+ }
1703
+ if (failNode === null) {
1704
+ child.fail = this.root;
1705
+ }
1706
+ else {
1707
+ child.fail = failNode.children.get(char);
1708
+ // Add outputs from the fail state to this node
1709
+ for (const output of child.fail.output) {
1710
+ child.output.add(output);
1711
+ }
1712
+ }
1713
+ }
1148
1714
  }
1715
+ this.built = true;
1149
1716
  }
1150
- wordsToCheck = wordsToCheck.filter((word) => !whitelist.includes(word.toLocaleLowerCase()));
1717
+ /**
1718
+ * Mencari semua kemunculan pola dalam teks
1719
+ * @param text Teks yang akan dicari
1720
+ * @returns Map pola yang ditemukan dengan jumlah kemunculannya
1721
+ */
1722
+ search(text) {
1723
+ if (!this.built) {
1724
+ this.build();
1725
+ }
1726
+ const matches = new Map();
1727
+ const normalizedText = text.toLowerCase();
1728
+ let node = this.root;
1729
+ for (let i = 0; i < normalizedText.length; i++) {
1730
+ const char = normalizedText[i];
1731
+ // Follow failure links until we find a matching transition or reach root
1732
+ while (node !== this.root && !node.children.has(char)) {
1733
+ node = node.fail;
1734
+ }
1735
+ // Try to follow the transition
1736
+ if (node.children.has(char)) {
1737
+ node = node.children.get(char);
1738
+ }
1739
+ // Check for any matches at this node
1740
+ for (const match of node.output) {
1741
+ matches.set(match, (matches.get(match) || 0) + 1);
1742
+ }
1743
+ }
1744
+ return matches;
1745
+ }
1746
+ /**
1747
+ * Mencari semua kemunculan pola dalam teks dan mengembalikan hanya pola unik
1748
+ * @param text Teks yang akan dicari
1749
+ * @returns Set pola yang ditemukan
1750
+ */
1751
+ searchUnique(text) {
1752
+ const matches = this.search(text);
1753
+ return new Set(matches.keys());
1754
+ }
1755
+ /**
1756
+ * Mengecek apakah teks mengandung setidaknya satu pola
1757
+ * @param text Teks yang akan dicari
1758
+ * @returns Boolean apakah pola ditemukan
1759
+ */
1760
+ containsAny(text) {
1761
+ if (!this.built) {
1762
+ this.build();
1763
+ }
1764
+ const normalizedText = text.toLowerCase();
1765
+ let node = this.root;
1766
+ for (let i = 0; i < normalizedText.length; i++) {
1767
+ const char = normalizedText[i];
1768
+ while (node !== this.root && !node.children.has(char)) {
1769
+ node = node.fail;
1770
+ }
1771
+ if (node.children.has(char)) {
1772
+ node = node.children.get(char);
1773
+ }
1774
+ if (node.output.size > 0) {
1775
+ return true;
1776
+ }
1777
+ }
1778
+ return false;
1779
+ }
1780
+ }
1781
+
1782
+ const globalAhoCorasick = new AhoCorasick();
1783
+ let ahoCorasickInitialized = false;
1784
+ function initializeAhoCorasick(words) {
1785
+ if (ahoCorasickInitialized)
1786
+ return;
1787
+ for (const word of words) {
1788
+ globalAhoCorasick.addPattern(word);
1789
+ }
1790
+ globalAhoCorasick.build();
1791
+ ahoCorasickInitialized = true;
1792
+ }
1793
+ function findProfanity(text, options = {}) {
1794
+ const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, useLevenshtein = false, maxLevenshteinDistance = 2, } = { ...DEFAULT_OPTIONS, ...options };
1795
+ const normalizedText = normalizeText(text);
1796
+ let baseWordsToCheck = wordList.length > 0 ? wordList : [];
1797
+ if (baseWordsToCheck.length === 0) {
1798
+ const filteredWords = wordObjects.filter((word) => {
1799
+ const matchCategory = categories
1800
+ ? categories.includes(word.category)
1801
+ : true;
1802
+ const matchRegion = regions ? regions.includes(word.region) : true;
1803
+ const matchSeverity = word.severity >= severityThreshold;
1804
+ return matchCategory && matchRegion && matchSeverity;
1805
+ });
1806
+ baseWordsToCheck = filteredWords.map((word) => word.word);
1807
+ }
1808
+ const aliasMap = new Map();
1809
+ wordObjects.forEach((wordObj) => {
1810
+ if (wordObj.aliases && wordObj.aliases.length > 0) {
1811
+ const matchCategory = categories
1812
+ ? categories.includes(wordObj.category)
1813
+ : true;
1814
+ const matchRegion = regions ? regions.includes(wordObj.region) : true;
1815
+ const matchSeverity = wordObj.severity >= severityThreshold;
1816
+ if (matchCategory && matchRegion && matchSeverity) {
1817
+ wordObj.aliases.forEach((alias) => {
1818
+ aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
1819
+ });
1820
+ }
1821
+ }
1822
+ });
1823
+ const wordsToCheck = [
1824
+ ...baseWordsToCheck,
1825
+ ...Array.from(aliasMap.keys()),
1826
+ ].filter((word) => !whitelist.includes(word.toLowerCase()));
1151
1827
  if (wordsToCheck.length === 0) {
1152
1828
  return [];
1153
1829
  }
1154
1830
  const matches = new Set();
1155
- wordsToCheck.forEach((word) => {
1156
- const regex = createWordRegex(word, {
1157
- wholeWord: !checkSubstring,
1158
- caseSensitive: false,
1159
- leetSpeak: false,
1160
- detectSplit: false,
1161
- indonesianVariation: false,
1162
- });
1163
- while ((regex.exec(normalizedText)) !== null) {
1164
- matches.add(word.toLowerCase());
1831
+ const actualMatches = new Map();
1832
+ initializeAhoCorasick(wordsToCheck);
1833
+ const basicMatches = globalAhoCorasick.searchUnique(normalizedText);
1834
+ for (const match of basicMatches) {
1835
+ const originalWord = aliasMap.get(match.toLowerCase()) || match.toLowerCase();
1836
+ matches.add(originalWord);
1837
+ if (!actualMatches.has(originalWord)) {
1838
+ actualMatches.set(originalWord, []);
1165
1839
  }
1166
- });
1840
+ actualMatches.get(originalWord)?.push(match);
1841
+ }
1167
1842
  if (detectLeetSpeak) {
1168
1843
  wordsToCheck.forEach((word) => {
1169
1844
  const leetRegex = createWordRegex(word, {
@@ -1173,8 +1848,14 @@ function findProfanity(text, options = {}) {
1173
1848
  detectSplit: false,
1174
1849
  indonesianVariation: false,
1175
1850
  });
1176
- while ((leetRegex.exec(text)) !== null) {
1177
- matches.add(word.toLowerCase());
1851
+ let match;
1852
+ while ((match = leetRegex.exec(text)) !== null) {
1853
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1854
+ matches.add(originalWord);
1855
+ if (!actualMatches.has(originalWord)) {
1856
+ actualMatches.set(originalWord, []);
1857
+ }
1858
+ actualMatches.get(originalWord)?.push(match[0]);
1178
1859
  }
1179
1860
  });
1180
1861
  }
@@ -1187,33 +1868,64 @@ function findProfanity(text, options = {}) {
1187
1868
  detectSplit: false,
1188
1869
  indonesianVariation: true,
1189
1870
  });
1190
- while ((variantRegex.exec(text)) !== null) {
1191
- matches.add(word.toLowerCase());
1871
+ let match;
1872
+ while ((match = variantRegex.exec(text)) !== null) {
1873
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1874
+ matches.add(originalWord);
1875
+ if (!actualMatches.has(originalWord)) {
1876
+ actualMatches.set(originalWord, []);
1877
+ }
1878
+ actualMatches.get(originalWord)?.push(match[0]);
1192
1879
  }
1193
1880
  });
1194
1881
  }
1195
1882
  if (detectSplit) {
1196
- if (detectSplitWords(text, wordsToCheck)) {
1197
- wordsToCheck.forEach((word) => {
1198
- const splitRegex = createWordRegex(word, {
1199
- wholeWord: false,
1200
- caseSensitive: false,
1201
- leetSpeak: false,
1202
- detectSplit: true,
1203
- indonesianVariation: false,
1204
- });
1205
- if (splitRegex.test(text)) {
1206
- matches.add(word.toLowerCase());
1207
- }
1883
+ wordsToCheck.forEach((word) => {
1884
+ const splitRegex = createWordRegex(word, {
1885
+ wholeWord: false,
1886
+ caseSensitive: false,
1887
+ leetSpeak: false,
1888
+ detectSplit: true,
1889
+ indonesianVariation: false,
1208
1890
  });
1209
- }
1891
+ let match;
1892
+ while ((match = splitRegex.exec(text)) !== null) {
1893
+ const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
1894
+ matches.add(originalWord);
1895
+ if (!actualMatches.has(originalWord)) {
1896
+ actualMatches.set(originalWord, []);
1897
+ }
1898
+ actualMatches.get(originalWord)?.push(match[0]);
1899
+ }
1900
+ });
1210
1901
  }
1211
1902
  if (detectSimilarity) {
1212
- const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
1213
- possibleProfanity.forEach((item) => {
1214
- matches.add(item.original.toLowerCase());
1215
- });
1903
+ if (useLevenshtein) {
1904
+ const possibleProfanity = findProfanityByLevenshteinDistance(text, wordsToCheck, similarityThreshold, maxLevenshteinDistance);
1905
+ possibleProfanity.forEach((item) => {
1906
+ const originalWord = aliasMap.get(item.original.toLowerCase()) ||
1907
+ item.original.toLowerCase();
1908
+ matches.add(originalWord);
1909
+ if (!actualMatches.has(originalWord)) {
1910
+ actualMatches.set(originalWord, []);
1911
+ }
1912
+ actualMatches.get(originalWord)?.push(item.word);
1913
+ });
1914
+ }
1915
+ else {
1916
+ const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
1917
+ possibleProfanity.forEach((item) => {
1918
+ matches.add(item.original.toLowerCase());
1919
+ const originalWord = aliasMap.get(item.original.toLowerCase()) ||
1920
+ item.original.toLowerCase();
1921
+ if (!actualMatches.has(originalWord)) {
1922
+ actualMatches.set(originalWord, []);
1923
+ }
1924
+ actualMatches.get(originalWord)?.push(item.word);
1925
+ });
1926
+ }
1216
1927
  }
1928
+ findProfanity.lastActualMatches = actualMatches;
1217
1929
  return Array.from(matches);
1218
1930
  }
1219
1931
  /**
@@ -1303,14 +2015,20 @@ function calculateSeverity(matchDetails) {
1303
2015
  * @returns FilterResult dengan hasil filter
1304
2016
  */
1305
2017
  function filter(text, options = {}) {
1306
- const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, } = { ...DEFAULT_OPTIONS, ...options };
2018
+ const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, detectSplit = false, detectSimilarity = false, useLevenshtein = false, maxLevenshteinDistance = 2, similarityThreshold = 0.8, } = { ...DEFAULT_OPTIONS, ...options };
1307
2019
  const matches = findProfanity(text, {
1308
2020
  ...options,
1309
2021
  detectLeetSpeak,
1310
2022
  whitelist,
1311
2023
  checkSubstring,
1312
2024
  indonesianVariation,
2025
+ detectSplit,
2026
+ detectSimilarity,
2027
+ useLevenshtein,
2028
+ maxLevenshteinDistance,
2029
+ similarityThreshold,
1313
2030
  });
2031
+ const actualMatches = findProfanity.lastActualMatches || new Map();
1314
2032
  const matchDetails = findProfanityWithMetadata(text, options);
1315
2033
  if (matches.length === 0) {
1316
2034
  return {
@@ -1325,36 +2043,120 @@ function filter(text, options = {}) {
1325
2043
  const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
1326
2044
  (m.aliases &&
1327
2045
  m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
1328
- const regex = createWordRegex(word, {
1329
- wholeWord: true,
1330
- caseSensitive: false,
1331
- leetSpeak: false,
1332
- detectSplit: false,
1333
- indonesianVariation: false,
2046
+ const variants = actualMatches.get(word.toLowerCase()) || [];
2047
+ variants.push(word);
2048
+ const uniqueVariants = [...new Set(variants)];
2049
+ uniqueVariants.forEach((variant) => {
2050
+ const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
2051
+ let match;
2052
+ while ((match = regex.exec(filteredText)) !== null) {
2053
+ const originalWord = match[0];
2054
+ if (whitelist.includes(originalWord.toLowerCase()))
2055
+ continue;
2056
+ let censoredWord;
2057
+ if (useRandomGrawlix) {
2058
+ censoredWord = makeRandomGrawlixString(originalWord.length);
2059
+ }
2060
+ else {
2061
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2062
+ }
2063
+ replacements.push({
2064
+ original: originalWord,
2065
+ censored: censoredWord,
2066
+ metadata,
2067
+ });
2068
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"), censoredWord);
2069
+ }
1334
2070
  });
1335
- let match;
1336
- const textToSearch = filteredText;
1337
- regex.lastIndex = 0;
1338
- while ((match = regex.exec(textToSearch)) !== null) {
1339
- const originalWord = match[0];
1340
- if (whitelist.includes(originalWord.toLowerCase()))
1341
- continue;
1342
- let censoredWord;
1343
- if (useRandomGrawlix) {
1344
- censoredWord = makeRandomGrawlixString(originalWord.length);
2071
+ if (detectSplit || detectLeetSpeak) {
2072
+ if (detectLeetSpeak) {
2073
+ const leetRegex = createWordRegex(word, {
2074
+ wholeWord: true,
2075
+ caseSensitive: false,
2076
+ leetSpeak: true,
2077
+ detectSplit: false,
2078
+ indonesianVariation: false,
2079
+ });
2080
+ let match;
2081
+ while ((match = leetRegex.exec(filteredText)) !== null) {
2082
+ const originalWord = match[0];
2083
+ if (whitelist.includes(originalWord.toLowerCase()))
2084
+ continue;
2085
+ let censoredWord;
2086
+ if (useRandomGrawlix) {
2087
+ censoredWord = makeRandomGrawlixString(originalWord.length);
2088
+ }
2089
+ else {
2090
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2091
+ }
2092
+ replacements.push({
2093
+ original: originalWord,
2094
+ censored: censoredWord,
2095
+ metadata,
2096
+ });
2097
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), "g"), censoredWord);
2098
+ }
1345
2099
  }
1346
- else {
1347
- censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2100
+ if (detectSplit) {
2101
+ const splitRegex = createWordRegex(word, {
2102
+ wholeWord: false,
2103
+ caseSensitive: false,
2104
+ leetSpeak: false,
2105
+ detectSplit: true,
2106
+ indonesianVariation: false,
2107
+ });
2108
+ let match;
2109
+ while ((match = splitRegex.exec(filteredText)) !== null) {
2110
+ const originalWord = match[0];
2111
+ if (whitelist.includes(originalWord.toLowerCase()))
2112
+ continue;
2113
+ let censoredWord;
2114
+ if (useRandomGrawlix) {
2115
+ censoredWord = makeRandomGrawlixString(originalWord.length);
2116
+ }
2117
+ else {
2118
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2119
+ }
2120
+ replacements.push({
2121
+ original: originalWord,
2122
+ censored: censoredWord,
2123
+ metadata,
2124
+ });
2125
+ filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), "g"), censoredWord);
2126
+ }
1348
2127
  }
1349
- replacements.push({
1350
- original: originalWord,
1351
- censored: censoredWord,
1352
- metadata,
1353
- });
1354
- const replaceRegex = new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g");
1355
- filteredText = filteredText.replace(replaceRegex, censoredWord);
1356
2128
  }
1357
2129
  });
2130
+ if (detectSimilarity && useLevenshtein) {
2131
+ matches.forEach((word) => {
2132
+ const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
2133
+ (m.aliases &&
2134
+ m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
2135
+ const variants = actualMatches.get(word.toLowerCase()) || [];
2136
+ variants.forEach((variant) => {
2137
+ const exactVariantRegex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
2138
+ let match;
2139
+ while ((match = exactVariantRegex.exec(filteredText)) !== null) {
2140
+ const originalWord = match[0];
2141
+ if (whitelist.includes(originalWord.toLowerCase()))
2142
+ continue;
2143
+ let censoredWord;
2144
+ if (useRandomGrawlix) {
2145
+ censoredWord = makeRandomGrawlixString(originalWord.length);
2146
+ }
2147
+ else {
2148
+ censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
2149
+ }
2150
+ replacements.push({
2151
+ original: originalWord,
2152
+ censored: censoredWord,
2153
+ metadata,
2154
+ });
2155
+ filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"), censoredWord);
2156
+ }
2157
+ });
2158
+ });
2159
+ }
1358
2160
  return {
1359
2161
  filtered: filteredText,
1360
2162
  censored: replacements.length,
@@ -1399,8 +2201,20 @@ function analyze(text, options = {}) {
1399
2201
  const severityScore = calculateSeverity(matchDetails);
1400
2202
  let similarWords = [];
1401
2203
  if (mergedOptions.detectSimilarity) {
1402
- const wordList = matchDetails.map((word) => word.word);
1403
- similarWords = findPossibleProfanityBySimiliarity(text, wordList, mergedOptions.similarityThreshold || 0.8);
2204
+ if (matchDetails.length > 0) {
2205
+ const wordList = matchDetails.map((word) => word.word);
2206
+ if (mergedOptions.useLevenshtein) {
2207
+ const levenshteinResults = findProfanityByLevenshteinDistance(text, wordList, mergedOptions.similarityThreshold || 0.8, mergedOptions.maxLevenshteinDistance || 2);
2208
+ similarWords = levenshteinResults.map((item) => ({
2209
+ word: item.word,
2210
+ original: item.original,
2211
+ similarity: item.similarity,
2212
+ }));
2213
+ }
2214
+ else {
2215
+ similarWords = findPossibleProfanityBySimiliarity(text, wordList, mergedOptions.similarityThreshold || 0.8);
2216
+ }
2217
+ }
1404
2218
  }
1405
2219
  return {
1406
2220
  hasProfanity: true,
@@ -1629,10 +2443,25 @@ class IDProfanityFilter {
1629
2443
  /**
1630
2444
  * Mengaktifkan deteksi berdasarkan kesamaan
1631
2445
  * @param threshold Threshold kesamaan (0-1)
2446
+ * @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
2447
+ * @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
2448
+ */
2449
+ enableSimilarityDetection(threshold = 0.8, useLevenshtein = false, maxLevenshteinDistance = 2) {
2450
+ this.options.detectSimilarity = true;
2451
+ this.options.similarityThreshold = threshold;
2452
+ this.options.useLevenshtein = useLevenshtein;
2453
+ this.options.maxLevenshteinDistance = maxLevenshteinDistance;
2454
+ }
2455
+ /**
2456
+ * Mengaktifkan deteksi berbasis Levenshtein distance
2457
+ * @param threshold Threshold kesamaan (0-1)
2458
+ * @param maxDistance Jarak maksimal Levenshtein (default: 2)
1632
2459
  */
1633
- enableSimilarityDetection(threshold = 0.8) {
2460
+ enableLevenshteinDetection(threshold = 0.8, maxDistance = 2) {
1634
2461
  this.options.detectSimilarity = true;
2462
+ this.options.useLevenshtein = true;
1635
2463
  this.options.similarityThreshold = threshold;
2464
+ this.options.maxLevenshteinDistance = maxDistance;
1636
2465
  }
1637
2466
  }
1638
2467
  const idFilter = {
@@ -1648,5 +2477,5 @@ const idFilter = {
1648
2477
  },
1649
2478
  };
1650
2479
 
1651
- export { CATEGORY_PRESETS, DEFAULT_OPTIONS, FILTER_PRESETS, IDProfanityFilter, REGION_PRESETS, REPLACEMENT_CHARS, addIndonesianVariations, addLeetSpeakVariations, addSplitVariations, analyze, analyzeBySentence, analyzeWithContext, batchAnalyze, calculateSeverity, censorWord, clusterSimilarWords, containsAnyWord, containsEuphemism, createContextRegex, createEvasionRegex, createIndonesianVariationRegex, createOptions, createWordFormRegex, createWordRegex, IDProfanityFilter as default, detectSplitWords, escapeRegExp, filter, findCategories, findMostSimilar, findPossibleProfanityBySimiliarity, findProfanity, findProfanityWithMetadata, findRegions, getContextAroundIndex, getPresetOptions, getRandomGrawlix, getReplacementChar, idFilter, isPossibleProfanityVariation, isProfane, levenshteinDistance, makeRandomGrawlixString, maskText, normalizeText, splitIntoSentences, stringSimilarity, toLeetSpeak };
2480
+ export { CATEGORY_PRESETS, DEFAULT_OPTIONS, FILTER_PRESETS, IDProfanityFilter, REGION_PRESETS, REPLACEMENT_CHARS, addIndonesianVariations, addLeetSpeakVariations, addSplitVariations, analyze, analyzeBySentence, analyzeWithContext, batchAnalyze, calculateSeverity, censorWord, clusterSimilarWords, containsAnyWord, containsEuphemism, createContextRegex, createEvasionRegex, createIndonesianVariationRegex, createOptions, createWordFormRegex, createWordRegex, IDProfanityFilter as default, detectSplitWords, escapeRegExp, filter, findCategories, findMostSimilar, findMostSimilarWithLevenshtein, findPossibleProfanityBySimiliarity, findProfanity, findProfanityByLevenshteinDistance, findProfanityWithMetadata, findRegions, getContextAroundIndex, getPresetOptions, getRandomGrawlix, getReplacementChar, idFilter, isPossibleProfanityVariation, isProfane, levenshteinDistance, makeRandomGrawlixString, maskText, normalizeText, splitIntoSentences, stringSimilarity, toLeetSpeak };
1652
2481
  //# sourceMappingURL=index.esm.js.map