@sideid/id-profanity-filter 1.10.6 → 1.11.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.eslintrc.js +44 -44
- package/.github/workflows/release.yml +62 -0
- package/CONTRIBUTING.md +150 -150
- package/LICENSE +21 -21
- package/README.md +548 -548
- package/dist/index.d.ts +989 -0
- package/dist/index.esm.js +545 -50
- package/dist/index.esm.js.map +1 -1
- package/dist/index.js +545 -50
- package/dist/index.js.map +1 -1
- package/dist/types/constants/categories/blasphemy.d.ts +4 -0
- package/dist/types/constants/categories/disgusting.d.ts +4 -0
- package/dist/types/constants/categories/drugs.d.ts +4 -0
- package/dist/types/constants/categories/profanity.d.ts +4 -0
- package/dist/types/constants/categories/slur.d.ts +4 -0
- package/dist/{constants → types/constants}/wordList.d.ts +1 -1
- package/dist/{core → types/core}/analyzer.d.ts +1 -1
- package/dist/{core → types/core}/filter.d.ts +1 -1
- package/dist/{core → types/core}/matcher.d.ts +1 -1
- package/dist/types/index.d.ts +375 -57
- package/dist/types/types/index.d.ts +59 -0
- package/dist/types/utils/ahoCorasick.d.ts +36 -0
- package/dist/{utils → types/utils}/similarityUtils.d.ts +10 -0
- package/eslint.config.mjs +40 -40
- package/examples/advanced.ts +120 -120
- package/examples/basic.ts +71 -71
- package/examples/custom-list.ts +140 -140
- package/jest.config.mjs +10 -10
- package/package.json +2 -1
- package/prettierrc +6 -6
- package/rollup.config.mjs +40 -35
- package/src/constants/categories/blasphemy.ts +25 -0
- package/src/constants/categories/disgusting.ts +82 -0
- package/src/constants/categories/drugs.ts +72 -0
- package/src/constants/categories/profanity.ts +139 -0
- package/src/constants/categories/slur.ts +102 -0
- package/src/constants/regions/general.ts +9 -0
- package/src/constants/regions/jawa.ts +356 -354
- package/src/core/analyzer.ts +28 -7
- package/src/core/matcher.ts +25 -18
- package/src/index.ts +15 -15
- package/src/utils/ahoCorasick.ts +179 -0
- package/src/utils/similarityUtils.ts +157 -20
- package/tsconfig.json +115 -115
- package/.github/workflows/ci.yml +0 -0
- package/dist/constants/categories/index.d.ts +0 -9
- package/dist/constants/regions/index.d.ts +0 -8
- package/test.js +0 -184
- /package/dist/{config → types/config}/options.d.ts +0 -0
- /package/dist/{constants → types/constants}/categories/insult.d.ts +0 -0
- /package/dist/{constants → types/constants}/categories/sexual.d.ts +0 -0
- /package/dist/{constants → types/constants}/regions/batak.d.ts +0 -0
- /package/dist/{constants → types/constants}/regions/betawi.d.ts +0 -0
- /package/dist/{constants → types/constants}/regions/general.d.ts +0 -0
- /package/dist/{constants → types/constants}/regions/jawa.d.ts +0 -0
- /package/dist/{constants → types/constants}/regions/sunda.d.ts +0 -0
- /package/dist/{utils → types/utils}/regexUtils.d.ts +0 -0
- /package/dist/{utils → types/utils}/stringUtils.d.ts +0 -0
package/dist/index.esm.js
CHANGED
|
@@ -197,6 +197,15 @@ const general = [
|
|
|
197
197
|
description: "Simbol yang sering digunakan untuk menandai konten pornografi",
|
|
198
198
|
context: "Digunakan untuk menandai konten seksual eksplisit",
|
|
199
199
|
},
|
|
200
|
+
{
|
|
201
|
+
word: "xnxx",
|
|
202
|
+
category: "sexual",
|
|
203
|
+
region: "general",
|
|
204
|
+
severity: 0.6,
|
|
205
|
+
aliases: ["xnxx", "xnx"],
|
|
206
|
+
description: "simbol yang sering digunakan untuk menandai konten pornografi",
|
|
207
|
+
context: "Digunakan untuk menandai konten seksual eksplisit",
|
|
208
|
+
}
|
|
200
209
|
];
|
|
201
210
|
general.map((item) => item.word);
|
|
202
211
|
|
|
@@ -214,7 +223,7 @@ const jawa = [
|
|
|
214
223
|
word: "jancok",
|
|
215
224
|
category: "sexual",
|
|
216
225
|
region: "jawa",
|
|
217
|
-
severity: 0.
|
|
226
|
+
severity: 0.9,
|
|
218
227
|
aliases: ["jancuk", "jncok", "jancuk", "jncuk", "dancok", "dancuk"],
|
|
219
228
|
description: "Kata umpatan kasar dalam Bahasa Jawa",
|
|
220
229
|
context: "Umpatan kasar yang umum digunakan di Jawa Timur",
|
|
@@ -250,7 +259,7 @@ const jawa = [
|
|
|
250
259
|
word: "mbokne ancok",
|
|
251
260
|
category: "insult",
|
|
252
261
|
region: "jawa",
|
|
253
|
-
severity: 0.
|
|
262
|
+
severity: 0.9,
|
|
254
263
|
aliases: ["mbokne", "mbokneancok"],
|
|
255
264
|
description: "Umpatan yang menyinggung ibu seseorang",
|
|
256
265
|
context: "Umpatan kasar yang menyinggung orangtua orang lain",
|
|
@@ -259,7 +268,7 @@ const jawa = [
|
|
|
259
268
|
word: "pekok",
|
|
260
269
|
category: "insult",
|
|
261
270
|
region: "jawa",
|
|
262
|
-
severity: 0.
|
|
271
|
+
severity: 0.7,
|
|
263
272
|
aliases: ["pekak", "pekilk"],
|
|
264
273
|
description: "Kata hinaan yang menunjukkan kebodohan",
|
|
265
274
|
context: "Hinaan untuk menyebut orang yang dianggap sangat bodoh",
|
|
@@ -291,6 +300,248 @@ const jawa = [
|
|
|
291
300
|
description: "Secara harfiah berarti kepalamu, digunakan sebagai umpatan ringan",
|
|
292
301
|
context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
|
|
293
302
|
},
|
|
303
|
+
{
|
|
304
|
+
word: "kontol",
|
|
305
|
+
category: "sexual",
|
|
306
|
+
region: "jawa",
|
|
307
|
+
severity: 0.8,
|
|
308
|
+
aliases: ["kntl", "kontl"],
|
|
309
|
+
description: "Mengacu ke alat kelamin laki-laki",
|
|
310
|
+
context: "Kata vulgar yang sering digunakan sebagai umpatan kasar",
|
|
311
|
+
},
|
|
312
|
+
{
|
|
313
|
+
word: "tempek",
|
|
314
|
+
category: "sexual",
|
|
315
|
+
region: "jawa",
|
|
316
|
+
severity: 0.8,
|
|
317
|
+
aliases: ["mpek", "torok", "tempk"],
|
|
318
|
+
description: "Mengacu pada alat kelamin perempuan",
|
|
319
|
+
context: "Kata vulgar yang digunakan sebagai umpatan atau hinaan",
|
|
320
|
+
},
|
|
321
|
+
{
|
|
322
|
+
word: "silit",
|
|
323
|
+
category: "insult",
|
|
324
|
+
region: "jawa",
|
|
325
|
+
severity: 0.6,
|
|
326
|
+
aliases: ["selet", "tilis"],
|
|
327
|
+
description: "Mengacu pada bagian dubur atau anus",
|
|
328
|
+
context: "Kata kasar yang digunakan sebagai hinaan",
|
|
329
|
+
},
|
|
330
|
+
{
|
|
331
|
+
word: "mbahmu",
|
|
332
|
+
category: "insult",
|
|
333
|
+
region: "jawa",
|
|
334
|
+
severity: 0.6,
|
|
335
|
+
aliases: ["mbahmu kiper", "mbah mu"],
|
|
336
|
+
description: "Hinaan yang menyinggung nenek/kakek seseorang",
|
|
337
|
+
context: "Umpatan ringan untuk menanggapi sesuatu yang tidak masuk akal",
|
|
338
|
+
},
|
|
339
|
+
{
|
|
340
|
+
word: "makmu",
|
|
341
|
+
category: "insult",
|
|
342
|
+
region: "jawa",
|
|
343
|
+
severity: 0.7,
|
|
344
|
+
aliases: ["mak mu", "mamamu"],
|
|
345
|
+
description: "Hinaan yang menyinggung ibu seseorang",
|
|
346
|
+
context: "Umpatan yang dianggap kasar karena menyinggung orang tua",
|
|
347
|
+
},
|
|
348
|
+
{
|
|
349
|
+
word: "bajingan",
|
|
350
|
+
category: "insult",
|
|
351
|
+
region: "jawa",
|
|
352
|
+
severity: 0.7,
|
|
353
|
+
aliases: [
|
|
354
|
+
"bajilak",
|
|
355
|
+
"bajhingan",
|
|
356
|
+
"bajingak",
|
|
357
|
+
"bajingseng",
|
|
358
|
+
"bajindul",
|
|
359
|
+
"bajigur",
|
|
360
|
+
"jingan",
|
|
361
|
+
],
|
|
362
|
+
description: "Sebutan untuk orang yang dianggap jahat atau tidak bermoral",
|
|
363
|
+
context: "Umpatan untuk mengekspresikan kemarahan atau kekesalan",
|
|
364
|
+
},
|
|
365
|
+
{
|
|
366
|
+
word: "cocote",
|
|
367
|
+
category: "insult",
|
|
368
|
+
region: "jawa",
|
|
369
|
+
severity: 0.6,
|
|
370
|
+
aliases: ["cocot", "bacot", "nyocot"],
|
|
371
|
+
description: "Mengacu pada mulut dengan konotasi negatif",
|
|
372
|
+
context: "Umpatan untuk menyuruh seseorang berhenti berbicara",
|
|
373
|
+
},
|
|
374
|
+
{
|
|
375
|
+
word: "ngentot",
|
|
376
|
+
category: "sexual",
|
|
377
|
+
region: "jawa",
|
|
378
|
+
severity: 0.9,
|
|
379
|
+
aliases: ["kentu", "kentot", "iclik", "ngtt", "iclk"],
|
|
380
|
+
description: "Mengacu pada aktivitas seksual",
|
|
381
|
+
context: "Kata vulgar yang digunakan sebagai umpatan kasar",
|
|
382
|
+
},
|
|
383
|
+
{
|
|
384
|
+
word: "edan",
|
|
385
|
+
category: "insult",
|
|
386
|
+
region: "jawa",
|
|
387
|
+
severity: 0.5,
|
|
388
|
+
aliases: ["gendeng", "gila", "gendheng", "sarap"],
|
|
389
|
+
description: "Secara harfiah berarti gila atau tidak waras",
|
|
390
|
+
context: "Umpatan untuk menyebut seseorang yang dianggap tidak masuk akal",
|
|
391
|
+
},
|
|
392
|
+
{
|
|
393
|
+
word: "dapuranmu",
|
|
394
|
+
category: "insult",
|
|
395
|
+
region: "jawa",
|
|
396
|
+
severity: 0.6,
|
|
397
|
+
aliases: ["raimu", "rai mu"],
|
|
398
|
+
description: "Secara harfiah mengacu pada wajah atau rupa seseorang",
|
|
399
|
+
context: "Umpatan untuk menghina penampilan atau wajah seseorang",
|
|
400
|
+
},
|
|
401
|
+
{
|
|
402
|
+
word: "damput",
|
|
403
|
+
category: "insult",
|
|
404
|
+
region: "jawa",
|
|
405
|
+
severity: 0.7,
|
|
406
|
+
aliases: ["diamput"],
|
|
407
|
+
description: "Variasi bentuk umpatan dengan makna serupa dengan jancok",
|
|
408
|
+
context: "Umpatan kasar untuk mengekspresikan kemarahan",
|
|
409
|
+
},
|
|
410
|
+
{
|
|
411
|
+
word: "mbathang",
|
|
412
|
+
category: "insult",
|
|
413
|
+
region: "jawa",
|
|
414
|
+
severity: 0.7,
|
|
415
|
+
aliases: ["mbatang"],
|
|
416
|
+
description: "Secara harfiah berarti bangkai",
|
|
417
|
+
context: "Umpatan kasar untuk menghina seseorang",
|
|
418
|
+
},
|
|
419
|
+
{
|
|
420
|
+
word: "ndlogok",
|
|
421
|
+
category: "insult",
|
|
422
|
+
region: "jawa",
|
|
423
|
+
severity: 0.6,
|
|
424
|
+
aliases: ["ndelodok", "ndlodok"],
|
|
425
|
+
description: "Mengacu pada tindakan yang dianggap bodoh atau tidak masuk akal",
|
|
426
|
+
context: "Hinaan untuk mengkritik tindakan seseorang",
|
|
427
|
+
},
|
|
428
|
+
{
|
|
429
|
+
word: "nggateli",
|
|
430
|
+
category: "insult",
|
|
431
|
+
region: "jawa",
|
|
432
|
+
severity: 0.5,
|
|
433
|
+
aliases: ["gateli", "gathel"],
|
|
434
|
+
description: "Secara harfiah berarti gatal atau menyebalkan",
|
|
435
|
+
context: "Ungkapan untuk menunjukkan kekesalan terhadap perilaku seseorang",
|
|
436
|
+
},
|
|
437
|
+
{
|
|
438
|
+
word: "perek",
|
|
439
|
+
category: "sexual",
|
|
440
|
+
region: "jawa",
|
|
441
|
+
severity: 0.8,
|
|
442
|
+
aliases: ["lonthe", "pelacur"],
|
|
443
|
+
description: "Istilah merendahkan untuk pekerja seks komersial",
|
|
444
|
+
context: "Kata kasar untuk menghina wanita",
|
|
445
|
+
},
|
|
446
|
+
{
|
|
447
|
+
word: "picek",
|
|
448
|
+
category: "insult",
|
|
449
|
+
region: "jawa",
|
|
450
|
+
severity: 0.6,
|
|
451
|
+
aliases: ["pcek", "buta"],
|
|
452
|
+
description: "Secara harfiah berarti buta atau tidak bisa melihat",
|
|
453
|
+
context: "Hinaan untuk orang yang dianggap tidak bisa melihat kenyataan",
|
|
454
|
+
},
|
|
455
|
+
{
|
|
456
|
+
word: "untumu",
|
|
457
|
+
category: "insult",
|
|
458
|
+
region: "jawa",
|
|
459
|
+
severity: 0.5,
|
|
460
|
+
aliases: ["gigimu", "untu mu"],
|
|
461
|
+
description: "Secara harfiah berarti gigimu",
|
|
462
|
+
context: "Umpatan ringan untuk menanggapi sesuatu yang tidak disetujui",
|
|
463
|
+
},
|
|
464
|
+
{
|
|
465
|
+
word: "goblog",
|
|
466
|
+
category: "insult",
|
|
467
|
+
region: "jawa",
|
|
468
|
+
severity: 0.7,
|
|
469
|
+
aliases: ["ghoblog", "goblok", "gobhlok", "pekok"],
|
|
470
|
+
description: "Kata hinaan yang menunjukkan kebodohan ekstrem",
|
|
471
|
+
context: "Hinaan untuk menyebut seseorang yang dianggap sangat bodoh",
|
|
472
|
+
},
|
|
473
|
+
{
|
|
474
|
+
word: "tolol",
|
|
475
|
+
category: "insult",
|
|
476
|
+
region: "jawa",
|
|
477
|
+
severity: 0.7,
|
|
478
|
+
aliases: ["tholol", "tlol"],
|
|
479
|
+
description: "Kata hinaan yang menunjukkan kebodohan",
|
|
480
|
+
context: "Hinaan untuk menyebut seseorang yang dianggap bodoh",
|
|
481
|
+
},
|
|
482
|
+
{
|
|
483
|
+
word: "budheg",
|
|
484
|
+
category: "insult",
|
|
485
|
+
region: "jawa",
|
|
486
|
+
severity: 0.6,
|
|
487
|
+
aliases: ["budeg", "bdeg"],
|
|
488
|
+
description: "Secara harfiah berarti tuli atau tidak bisa mendengar",
|
|
489
|
+
context: "Hinaan untuk orang yang dianggap tidak mau mendengarkan",
|
|
490
|
+
},
|
|
491
|
+
{
|
|
492
|
+
word: "jiangkrik",
|
|
493
|
+
category: "insult",
|
|
494
|
+
region: "jawa",
|
|
495
|
+
severity: 0.4,
|
|
496
|
+
aliases: ["jiangkrek", "jangkrik"],
|
|
497
|
+
description: "Secara harfiah berarti jangkrik, digunakan sebagai eufemisme",
|
|
498
|
+
context: "Umpatan ringan sebagai pengganti kata kasar yang lebih vulgar",
|
|
499
|
+
},
|
|
500
|
+
{
|
|
501
|
+
word: "diamput",
|
|
502
|
+
category: "insult",
|
|
503
|
+
region: "jawa",
|
|
504
|
+
severity: 0.8,
|
|
505
|
+
aliases: ["damput", "djamput"],
|
|
506
|
+
description: "Bentuk umpatan kasar dengan makna serupa jancok",
|
|
507
|
+
context: "Kata kasar untuk mengekspresikan kemarahan",
|
|
508
|
+
},
|
|
509
|
+
{
|
|
510
|
+
word: "celeng",
|
|
511
|
+
category: "insult",
|
|
512
|
+
region: "jawa",
|
|
513
|
+
severity: 0.6,
|
|
514
|
+
aliases: ["cleng", "babi hutan"],
|
|
515
|
+
description: "Secara harfiah berarti babi hutan",
|
|
516
|
+
context: "Hinaan untuk orang yang dianggap jorok atau rakus",
|
|
517
|
+
},
|
|
518
|
+
{
|
|
519
|
+
word: "kampret",
|
|
520
|
+
category: "insult",
|
|
521
|
+
region: "jawa",
|
|
522
|
+
severity: 0.5,
|
|
523
|
+
aliases: ["kmpret", "kmprt"],
|
|
524
|
+
description: "Secara harfiah berarti kelelawar kecil",
|
|
525
|
+
context: "Umpatan ringan untuk mengekspresikan kekesalan",
|
|
526
|
+
},
|
|
527
|
+
{
|
|
528
|
+
word: "ndeso",
|
|
529
|
+
category: "insult",
|
|
530
|
+
region: "jawa",
|
|
531
|
+
severity: 0.4,
|
|
532
|
+
aliases: ["ndesa", "deso"],
|
|
533
|
+
description: "Secara harfiah berarti dari desa atau kampungan",
|
|
534
|
+
context: "Hinaan untuk orang yang dianggap kurang modern atau berpendidikan",
|
|
535
|
+
},
|
|
536
|
+
{
|
|
537
|
+
word: "kere",
|
|
538
|
+
category: "insult",
|
|
539
|
+
region: "jawa",
|
|
540
|
+
severity: 0.5,
|
|
541
|
+
aliases: ["miskin", "mlarat"],
|
|
542
|
+
description: "Secara harfiah berarti miskin atau tidak punya uang",
|
|
543
|
+
context: "Hinaan untuk status ekonomi seseorang yang dianggap rendah",
|
|
544
|
+
},
|
|
294
545
|
{
|
|
295
546
|
word: "itil",
|
|
296
547
|
category: "sexual",
|
|
@@ -1084,29 +1335,73 @@ function clusterSimilarWords(words, threshold = 0.8) {
|
|
|
1084
1335
|
}
|
|
1085
1336
|
/**
|
|
1086
1337
|
* Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
|
|
1338
|
+
* dengan optimasi untuk mengurangi kompleksitas
|
|
1087
1339
|
*
|
|
1088
1340
|
* @param text Teks yang akan diperiksa
|
|
1089
1341
|
* @param profanityWords Daftar kata kotor
|
|
1090
1342
|
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
1091
1343
|
* @returns Array kata yang mungkin merupakan kata kotor
|
|
1092
1344
|
*/
|
|
1345
|
+
/**
|
|
1346
|
+
* Cari kata-kata kotor yang mungkin dari teks berdasarkan kemiripan string
|
|
1347
|
+
* dengan optimasi biar prosesnya nggak terlalu berat
|
|
1348
|
+
*
|
|
1349
|
+
* @param text Teks yang mau dicek
|
|
1350
|
+
* @param profanityWords Daftar kata-kata kotor/kasar
|
|
1351
|
+
* @param threshold Batas minimal kemiripan (default: 0.8)
|
|
1352
|
+
* @returns Array kata yang kemungkinan kata kotor/kasar
|
|
1353
|
+
*/
|
|
1093
1354
|
function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.8) {
|
|
1094
1355
|
const result = [];
|
|
1356
|
+
// Optimasi dengan bikin map kata-kata kotor berdasarkan huruf pertama
|
|
1357
|
+
const profanityMap = new Map();
|
|
1358
|
+
// Kelompokkan kata-kata kotor berdasarkan huruf pertama biar pencariannya lebih cepat
|
|
1359
|
+
for (const word of profanityWords) {
|
|
1360
|
+
if (word.length < 1)
|
|
1361
|
+
continue;
|
|
1362
|
+
const firstChar = word[0].toLowerCase();
|
|
1363
|
+
if (!profanityMap.has(firstChar)) {
|
|
1364
|
+
profanityMap.set(firstChar, []);
|
|
1365
|
+
}
|
|
1366
|
+
profanityMap.get(firstChar).push(word);
|
|
1367
|
+
}
|
|
1095
1368
|
const words = text.toLowerCase().split(/\s+/);
|
|
1096
1369
|
for (const word of words) {
|
|
1097
1370
|
if (word.length < 3)
|
|
1098
1371
|
continue;
|
|
1099
|
-
|
|
1372
|
+
// Cuma bandingin dengan kata-kata kotor yang huruf pertamanya sama
|
|
1373
|
+
// atau yang perbedaan panjangnya masih masuk akal
|
|
1374
|
+
const firstChar = word[0];
|
|
1375
|
+
const candidateWords = profanityMap.get(firstChar) || [];
|
|
1376
|
+
// Cek juga huruf-huruf yang berdekatan (buat antisipasi typo di huruf pertama)
|
|
1377
|
+
// Ini opsional tapi bikin deteksinya lebih bagus
|
|
1378
|
+
const charCode = firstChar.charCodeAt(0);
|
|
1379
|
+
const prevChar = String.fromCharCode(charCode - 1);
|
|
1380
|
+
const nextChar = String.fromCharCode(charCode + 1);
|
|
1381
|
+
const adjacentCandidates = [
|
|
1382
|
+
...(profanityMap.get(prevChar) || []),
|
|
1383
|
+
...(profanityMap.get(nextChar) || []),
|
|
1384
|
+
];
|
|
1385
|
+
// Gabungin kandidat-kandidatnya, prioritasin yang huruf pertamanya sama persis
|
|
1386
|
+
const allCandidates = [...candidateWords, ...adjacentCandidates];
|
|
1387
|
+
// Filter kandidat berdasarkan perbedaan panjang sebelum ngitung kemiripannya
|
|
1388
|
+
const lengthFilteredCandidates = allCandidates.filter((candidate) => Math.abs(candidate.length - word.length) <= 2);
|
|
1389
|
+
// Cari yang paling cocok
|
|
1390
|
+
let bestMatch = null;
|
|
1391
|
+
for (const profanity of lengthFilteredCandidates) {
|
|
1100
1392
|
const similarity = stringSimilarity(word, profanity);
|
|
1101
|
-
if (similarity >= threshold
|
|
1102
|
-
|
|
1393
|
+
if (similarity >= threshold &&
|
|
1394
|
+
(!bestMatch || similarity > bestMatch.similarity)) {
|
|
1395
|
+
bestMatch = {
|
|
1103
1396
|
word,
|
|
1104
1397
|
original: profanity,
|
|
1105
1398
|
similarity,
|
|
1106
|
-
}
|
|
1107
|
-
break;
|
|
1399
|
+
};
|
|
1108
1400
|
}
|
|
1109
1401
|
}
|
|
1402
|
+
if (bestMatch) {
|
|
1403
|
+
result.push(bestMatch);
|
|
1404
|
+
}
|
|
1110
1405
|
}
|
|
1111
1406
|
return result;
|
|
1112
1407
|
}
|
|
@@ -1121,30 +1416,75 @@ function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.
|
|
|
1121
1416
|
*/
|
|
1122
1417
|
function findProfanityByLevenshteinDistance(text, profanityWords, threshold = 0.8, maxDistance = 2) {
|
|
1123
1418
|
const result = [];
|
|
1419
|
+
// map kata-kata kotor dikelompokkan sesuai panjangnya
|
|
1420
|
+
const profanityByLength = new Map();
|
|
1421
|
+
// Kelompokkin kata-kata kotor berdasarkan panjangnya biar pencarian lebih cepat
|
|
1422
|
+
for (const word of profanityWords) {
|
|
1423
|
+
const length = word.length;
|
|
1424
|
+
if (!profanityByLength.has(length)) {
|
|
1425
|
+
profanityByLength.set(length, []);
|
|
1426
|
+
}
|
|
1427
|
+
profanityByLength.get(length).push(word);
|
|
1428
|
+
}
|
|
1124
1429
|
const words = text.toLowerCase().split(/\s+/);
|
|
1125
1430
|
for (const word of words) {
|
|
1126
1431
|
if (word.length < 3)
|
|
1127
1432
|
continue;
|
|
1128
|
-
|
|
1129
|
-
|
|
1130
|
-
|
|
1131
|
-
const
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1433
|
+
let bestMatch = null;
|
|
1434
|
+
for (let len = Math.max(3, word.length - maxDistance); len <= word.length + maxDistance; len++) {
|
|
1435
|
+
const candidates = profanityByLength.get(len) || [];
|
|
1436
|
+
for (const profanity of candidates) {
|
|
1437
|
+
if (!isCharacterCountSimilar(word, profanity, maxDistance)) {
|
|
1438
|
+
continue;
|
|
1439
|
+
}
|
|
1440
|
+
const distance = levenshteinDistance(word, profanity);
|
|
1441
|
+
if (distance <= maxDistance) {
|
|
1442
|
+
const similarity = 1 - distance / Math.max(word.length, profanity.length);
|
|
1443
|
+
if (similarity >= threshold &&
|
|
1444
|
+
(!bestMatch || similarity > bestMatch.similarity)) {
|
|
1445
|
+
bestMatch = {
|
|
1446
|
+
word,
|
|
1447
|
+
original: profanity,
|
|
1448
|
+
similarity,
|
|
1449
|
+
distance,
|
|
1450
|
+
};
|
|
1451
|
+
if (distance === 0 || similarity > 0.95) {
|
|
1452
|
+
break;
|
|
1453
|
+
}
|
|
1454
|
+
}
|
|
1142
1455
|
}
|
|
1143
1456
|
}
|
|
1144
1457
|
}
|
|
1458
|
+
if (bestMatch) {
|
|
1459
|
+
result.push(bestMatch);
|
|
1460
|
+
}
|
|
1145
1461
|
}
|
|
1146
1462
|
return result;
|
|
1147
1463
|
}
|
|
1464
|
+
/**
|
|
1465
|
+
* Helper function to efficiently check if character counts between two strings
|
|
1466
|
+
* are similar enough to warrant a full Levenshtein calculation
|
|
1467
|
+
*/
|
|
1468
|
+
function isCharacterCountSimilar(str1, str2, maxDifference) {
|
|
1469
|
+
const charCount1 = {};
|
|
1470
|
+
const charCount2 = {};
|
|
1471
|
+
for (const char of str1) {
|
|
1472
|
+
charCount1[char] = (charCount1[char] || 0) + 1;
|
|
1473
|
+
}
|
|
1474
|
+
for (const char of str2) {
|
|
1475
|
+
charCount2[char] = (charCount2[char] || 0) + 1;
|
|
1476
|
+
}
|
|
1477
|
+
let diffCount = 0;
|
|
1478
|
+
for (const char in charCount1) {
|
|
1479
|
+
diffCount += Math.abs((charCount1[char] || 0) - (charCount2[char] || 0));
|
|
1480
|
+
}
|
|
1481
|
+
for (const char in charCount2) {
|
|
1482
|
+
if (!charCount1[char]) {
|
|
1483
|
+
diffCount += charCount2[char];
|
|
1484
|
+
}
|
|
1485
|
+
}
|
|
1486
|
+
return diffCount <= maxDifference * 2;
|
|
1487
|
+
}
|
|
1148
1488
|
|
|
1149
1489
|
const DEFAULT_OPTIONS = {
|
|
1150
1490
|
replaceWith: "*",
|
|
@@ -1299,6 +1639,157 @@ function makeRandomGrawlixString(length) {
|
|
|
1299
1639
|
return result;
|
|
1300
1640
|
}
|
|
1301
1641
|
|
|
1642
|
+
/**
|
|
1643
|
+
* Implementasi algoritma Aho-Corasick untuk pencocokan string
|
|
1644
|
+
* Algoritma ini memungkinkan pencarian banyak pola dalam satu kali pembacaan teks
|
|
1645
|
+
*/
|
|
1646
|
+
class AhoCorasick {
|
|
1647
|
+
constructor() {
|
|
1648
|
+
this.built = false;
|
|
1649
|
+
this.root = {
|
|
1650
|
+
children: new Map(),
|
|
1651
|
+
fail: null,
|
|
1652
|
+
output: new Set(),
|
|
1653
|
+
depth: 0,
|
|
1654
|
+
};
|
|
1655
|
+
}
|
|
1656
|
+
/**
|
|
1657
|
+
* Menambahkan pola ke dalam trie
|
|
1658
|
+
* @param pattern Pola yang akan ditambahkan
|
|
1659
|
+
*/
|
|
1660
|
+
addPattern(pattern) {
|
|
1661
|
+
if (this.built) {
|
|
1662
|
+
throw new Error("Cannot add patterns after the automaton is built");
|
|
1663
|
+
}
|
|
1664
|
+
let node = this.root;
|
|
1665
|
+
const normalizedPattern = pattern.toLowerCase();
|
|
1666
|
+
for (let i = 0; i < normalizedPattern.length; i++) {
|
|
1667
|
+
const char = normalizedPattern[i];
|
|
1668
|
+
if (!node.children.has(char)) {
|
|
1669
|
+
node.children.set(char, {
|
|
1670
|
+
children: new Map(),
|
|
1671
|
+
fail: null,
|
|
1672
|
+
output: new Set(),
|
|
1673
|
+
depth: node.depth + 1,
|
|
1674
|
+
char,
|
|
1675
|
+
});
|
|
1676
|
+
}
|
|
1677
|
+
node = node.children.get(char);
|
|
1678
|
+
}
|
|
1679
|
+
node.output.add(normalizedPattern);
|
|
1680
|
+
}
|
|
1681
|
+
/**
|
|
1682
|
+
* Membangun fungsi failure
|
|
1683
|
+
*/
|
|
1684
|
+
build() {
|
|
1685
|
+
if (this.built)
|
|
1686
|
+
return;
|
|
1687
|
+
const queue = [];
|
|
1688
|
+
// Set fail pointer for depth 1 nodes to root
|
|
1689
|
+
for (const child of this.root.children.values()) {
|
|
1690
|
+
child.fail = this.root;
|
|
1691
|
+
queue.push(child);
|
|
1692
|
+
}
|
|
1693
|
+
// BFS to build failure links
|
|
1694
|
+
while (queue.length > 0) {
|
|
1695
|
+
const current = queue.shift();
|
|
1696
|
+
for (const [char, child] of current.children.entries()) {
|
|
1697
|
+
queue.push(child);
|
|
1698
|
+
let failNode = current.fail;
|
|
1699
|
+
// Find the longest proper suffix that is also a prefix
|
|
1700
|
+
while (failNode !== null && !failNode.children.has(char)) {
|
|
1701
|
+
failNode = failNode.fail;
|
|
1702
|
+
}
|
|
1703
|
+
if (failNode === null) {
|
|
1704
|
+
child.fail = this.root;
|
|
1705
|
+
}
|
|
1706
|
+
else {
|
|
1707
|
+
child.fail = failNode.children.get(char);
|
|
1708
|
+
// Add outputs from the fail state to this node
|
|
1709
|
+
for (const output of child.fail.output) {
|
|
1710
|
+
child.output.add(output);
|
|
1711
|
+
}
|
|
1712
|
+
}
|
|
1713
|
+
}
|
|
1714
|
+
}
|
|
1715
|
+
this.built = true;
|
|
1716
|
+
}
|
|
1717
|
+
/**
|
|
1718
|
+
* Mencari semua kemunculan pola dalam teks
|
|
1719
|
+
* @param text Teks yang akan dicari
|
|
1720
|
+
* @returns Map pola yang ditemukan dengan jumlah kemunculannya
|
|
1721
|
+
*/
|
|
1722
|
+
search(text) {
|
|
1723
|
+
if (!this.built) {
|
|
1724
|
+
this.build();
|
|
1725
|
+
}
|
|
1726
|
+
const matches = new Map();
|
|
1727
|
+
const normalizedText = text.toLowerCase();
|
|
1728
|
+
let node = this.root;
|
|
1729
|
+
for (let i = 0; i < normalizedText.length; i++) {
|
|
1730
|
+
const char = normalizedText[i];
|
|
1731
|
+
// Follow failure links until we find a matching transition or reach root
|
|
1732
|
+
while (node !== this.root && !node.children.has(char)) {
|
|
1733
|
+
node = node.fail;
|
|
1734
|
+
}
|
|
1735
|
+
// Try to follow the transition
|
|
1736
|
+
if (node.children.has(char)) {
|
|
1737
|
+
node = node.children.get(char);
|
|
1738
|
+
}
|
|
1739
|
+
// Check for any matches at this node
|
|
1740
|
+
for (const match of node.output) {
|
|
1741
|
+
matches.set(match, (matches.get(match) || 0) + 1);
|
|
1742
|
+
}
|
|
1743
|
+
}
|
|
1744
|
+
return matches;
|
|
1745
|
+
}
|
|
1746
|
+
/**
|
|
1747
|
+
* Mencari semua kemunculan pola dalam teks dan mengembalikan hanya pola unik
|
|
1748
|
+
* @param text Teks yang akan dicari
|
|
1749
|
+
* @returns Set pola yang ditemukan
|
|
1750
|
+
*/
|
|
1751
|
+
searchUnique(text) {
|
|
1752
|
+
const matches = this.search(text);
|
|
1753
|
+
return new Set(matches.keys());
|
|
1754
|
+
}
|
|
1755
|
+
/**
|
|
1756
|
+
* Mengecek apakah teks mengandung setidaknya satu pola
|
|
1757
|
+
* @param text Teks yang akan dicari
|
|
1758
|
+
* @returns Boolean apakah pola ditemukan
|
|
1759
|
+
*/
|
|
1760
|
+
containsAny(text) {
|
|
1761
|
+
if (!this.built) {
|
|
1762
|
+
this.build();
|
|
1763
|
+
}
|
|
1764
|
+
const normalizedText = text.toLowerCase();
|
|
1765
|
+
let node = this.root;
|
|
1766
|
+
for (let i = 0; i < normalizedText.length; i++) {
|
|
1767
|
+
const char = normalizedText[i];
|
|
1768
|
+
while (node !== this.root && !node.children.has(char)) {
|
|
1769
|
+
node = node.fail;
|
|
1770
|
+
}
|
|
1771
|
+
if (node.children.has(char)) {
|
|
1772
|
+
node = node.children.get(char);
|
|
1773
|
+
}
|
|
1774
|
+
if (node.output.size > 0) {
|
|
1775
|
+
return true;
|
|
1776
|
+
}
|
|
1777
|
+
}
|
|
1778
|
+
return false;
|
|
1779
|
+
}
|
|
1780
|
+
}
|
|
1781
|
+
|
|
1782
|
+
const globalAhoCorasick = new AhoCorasick();
|
|
1783
|
+
let ahoCorasickInitialized = false;
|
|
1784
|
+
function initializeAhoCorasick(words) {
|
|
1785
|
+
if (ahoCorasickInitialized)
|
|
1786
|
+
return;
|
|
1787
|
+
for (const word of words) {
|
|
1788
|
+
globalAhoCorasick.addPattern(word);
|
|
1789
|
+
}
|
|
1790
|
+
globalAhoCorasick.build();
|
|
1791
|
+
ahoCorasickInitialized = true;
|
|
1792
|
+
}
|
|
1302
1793
|
function findProfanity(text, options = {}) {
|
|
1303
1794
|
const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, useLevenshtein = false, maxLevenshteinDistance = 2, } = { ...DEFAULT_OPTIONS, ...options };
|
|
1304
1795
|
const normalizedText = normalizeText(text);
|
|
@@ -1338,24 +1829,16 @@ function findProfanity(text, options = {}) {
|
|
|
1338
1829
|
}
|
|
1339
1830
|
const matches = new Set();
|
|
1340
1831
|
const actualMatches = new Map();
|
|
1341
|
-
wordsToCheck
|
|
1342
|
-
|
|
1343
|
-
|
|
1344
|
-
|
|
1345
|
-
|
|
1346
|
-
|
|
1347
|
-
|
|
1348
|
-
});
|
|
1349
|
-
let match;
|
|
1350
|
-
while ((match = regex.exec(normalizedText)) !== null) {
|
|
1351
|
-
const originalWord = aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
1352
|
-
matches.add(originalWord);
|
|
1353
|
-
if (!actualMatches.has(originalWord)) {
|
|
1354
|
-
actualMatches.set(originalWord, []);
|
|
1355
|
-
}
|
|
1356
|
-
actualMatches.get(originalWord)?.push(match[0]);
|
|
1832
|
+
initializeAhoCorasick(wordsToCheck);
|
|
1833
|
+
const basicMatches = globalAhoCorasick.searchUnique(normalizedText);
|
|
1834
|
+
for (const match of basicMatches) {
|
|
1835
|
+
const originalWord = aliasMap.get(match.toLowerCase()) || match.toLowerCase();
|
|
1836
|
+
matches.add(originalWord);
|
|
1837
|
+
if (!actualMatches.has(originalWord)) {
|
|
1838
|
+
actualMatches.set(originalWord, []);
|
|
1357
1839
|
}
|
|
1358
|
-
|
|
1840
|
+
actualMatches.get(originalWord)?.push(match);
|
|
1841
|
+
}
|
|
1359
1842
|
if (detectLeetSpeak) {
|
|
1360
1843
|
wordsToCheck.forEach((word) => {
|
|
1361
1844
|
const leetRegex = createWordRegex(word, {
|
|
@@ -1532,7 +2015,7 @@ function calculateSeverity(matchDetails) {
|
|
|
1532
2015
|
* @returns FilterResult dengan hasil filter
|
|
1533
2016
|
*/
|
|
1534
2017
|
function filter(text, options = {}) {
|
|
1535
|
-
const { replaceWith =
|
|
2018
|
+
const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, detectSplit = false, detectSimilarity = false, useLevenshtein = false, maxLevenshteinDistance = 2, similarityThreshold = 0.8, } = { ...DEFAULT_OPTIONS, ...options };
|
|
1536
2019
|
const matches = findProfanity(text, {
|
|
1537
2020
|
...options,
|
|
1538
2021
|
detectLeetSpeak,
|
|
@@ -1564,7 +2047,7 @@ function filter(text, options = {}) {
|
|
|
1564
2047
|
variants.push(word);
|
|
1565
2048
|
const uniqueVariants = [...new Set(variants)];
|
|
1566
2049
|
uniqueVariants.forEach((variant) => {
|
|
1567
|
-
const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`,
|
|
2050
|
+
const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
|
|
1568
2051
|
let match;
|
|
1569
2052
|
while ((match = regex.exec(filteredText)) !== null) {
|
|
1570
2053
|
const originalWord = match[0];
|
|
@@ -1582,7 +2065,7 @@ function filter(text, options = {}) {
|
|
|
1582
2065
|
censored: censoredWord,
|
|
1583
2066
|
metadata,
|
|
1584
2067
|
});
|
|
1585
|
-
filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`,
|
|
2068
|
+
filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"), censoredWord);
|
|
1586
2069
|
}
|
|
1587
2070
|
});
|
|
1588
2071
|
if (detectSplit || detectLeetSpeak) {
|
|
@@ -1611,7 +2094,7 @@ function filter(text, options = {}) {
|
|
|
1611
2094
|
censored: censoredWord,
|
|
1612
2095
|
metadata,
|
|
1613
2096
|
});
|
|
1614
|
-
filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord),
|
|
2097
|
+
filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), "g"), censoredWord);
|
|
1615
2098
|
}
|
|
1616
2099
|
}
|
|
1617
2100
|
if (detectSplit) {
|
|
@@ -1639,7 +2122,7 @@ function filter(text, options = {}) {
|
|
|
1639
2122
|
censored: censoredWord,
|
|
1640
2123
|
metadata,
|
|
1641
2124
|
});
|
|
1642
|
-
filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord),
|
|
2125
|
+
filteredText = filteredText.replace(new RegExp(escapeRegExp(originalWord), "g"), censoredWord);
|
|
1643
2126
|
}
|
|
1644
2127
|
}
|
|
1645
2128
|
}
|
|
@@ -1651,7 +2134,7 @@ function filter(text, options = {}) {
|
|
|
1651
2134
|
m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
|
|
1652
2135
|
const variants = actualMatches.get(word.toLowerCase()) || [];
|
|
1653
2136
|
variants.forEach((variant) => {
|
|
1654
|
-
const exactVariantRegex = new RegExp(`\\b${escapeRegExp(variant)}\\b`,
|
|
2137
|
+
const exactVariantRegex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
|
|
1655
2138
|
let match;
|
|
1656
2139
|
while ((match = exactVariantRegex.exec(filteredText)) !== null) {
|
|
1657
2140
|
const originalWord = match[0];
|
|
@@ -1669,7 +2152,7 @@ function filter(text, options = {}) {
|
|
|
1669
2152
|
censored: censoredWord,
|
|
1670
2153
|
metadata,
|
|
1671
2154
|
});
|
|
1672
|
-
filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`,
|
|
2155
|
+
filteredText = filteredText.replace(new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"), censoredWord);
|
|
1673
2156
|
}
|
|
1674
2157
|
});
|
|
1675
2158
|
});
|
|
@@ -1718,8 +2201,20 @@ function analyze(text, options = {}) {
|
|
|
1718
2201
|
const severityScore = calculateSeverity(matchDetails);
|
|
1719
2202
|
let similarWords = [];
|
|
1720
2203
|
if (mergedOptions.detectSimilarity) {
|
|
1721
|
-
|
|
1722
|
-
|
|
2204
|
+
if (matchDetails.length > 0) {
|
|
2205
|
+
const wordList = matchDetails.map((word) => word.word);
|
|
2206
|
+
if (mergedOptions.useLevenshtein) {
|
|
2207
|
+
const levenshteinResults = findProfanityByLevenshteinDistance(text, wordList, mergedOptions.similarityThreshold || 0.8, mergedOptions.maxLevenshteinDistance || 2);
|
|
2208
|
+
similarWords = levenshteinResults.map((item) => ({
|
|
2209
|
+
word: item.word,
|
|
2210
|
+
original: item.original,
|
|
2211
|
+
similarity: item.similarity,
|
|
2212
|
+
}));
|
|
2213
|
+
}
|
|
2214
|
+
else {
|
|
2215
|
+
similarWords = findPossibleProfanityBySimiliarity(text, wordList, mergedOptions.similarityThreshold || 0.8);
|
|
2216
|
+
}
|
|
2217
|
+
}
|
|
1723
2218
|
}
|
|
1724
2219
|
return {
|
|
1725
2220
|
hasProfanity: true,
|
|
@@ -1816,9 +2311,9 @@ function analyzeWithContext(text, contextWindowSize = 5, options = {}) {
|
|
|
1816
2311
|
const regex = createContextRegex(word, contextWindowSize);
|
|
1817
2312
|
let match;
|
|
1818
2313
|
while ((match = regex.exec(text)) !== null) {
|
|
1819
|
-
const beforeContext = match[1] ||
|
|
2314
|
+
const beforeContext = match[1] || "";
|
|
1820
2315
|
const wordMatch = match[2];
|
|
1821
|
-
const afterContext = match[3] ||
|
|
2316
|
+
const afterContext = match[3] || "";
|
|
1822
2317
|
result.push({
|
|
1823
2318
|
word: wordMatch,
|
|
1824
2319
|
context: beforeContext + wordMatch + afterContext,
|