@sideid/id-profanity-filter 1.9.5 → 1.10.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.eslintrc.js +30 -2
- package/README.md +272 -9
- package/dist/config/options.d.ts +24 -0
- package/dist/constants/categories/index.d.ts +9 -0
- package/dist/constants/regions/index.d.ts +8 -0
- package/dist/constants/wordList.d.ts +1 -1
- package/dist/core/analyzer.d.ts +1 -1
- package/dist/core/filter.d.ts +1 -1
- package/dist/core/matcher.d.ts +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.esm.js +411 -77
- package/dist/index.esm.js.map +1 -1
- package/dist/index.js +412 -76
- package/dist/index.js.map +1 -1
- package/dist/types/index.d.ts +2 -0
- package/dist/utils/similarityUtils.d.ts +25 -0
- package/eslint.config.mjs +40 -0
- package/examples/advanced.ts +120 -0
- package/examples/basic.ts +60 -41
- package/examples/custom-list.ts +140 -0
- package/package.json +2 -2
- package/src/config/options.ts +2 -0
- package/src/constants/regions/general.ts +102 -2
- package/src/constants/regions/jawa.ts +10 -0
- package/src/constants/wordList.ts +21 -14
- package/src/core/analyzer.ts +8 -14
- package/src/core/filter.ts +182 -41
- package/src/core/matcher.ts +131 -61
- package/src/index.ts +34 -15
- package/src/types/index.ts +4 -2
- package/src/utils/regexUtils.ts +0 -1
- package/src/utils/similarityUtils.ts +97 -2
- package/test.js +184 -0
- package/src/constants/categories/index.ts +0 -31
- package/src/constants/regions/index.ts +0 -62
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
import { IDProfanityFilter, idFilter, FilterOptions } from '../src/index';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Contoh Penggunaan ID-Profanity-Filter dengan Daftar Kustom
|
|
5
|
+
* =======================================================
|
|
6
|
+
* File ini menunjukkan cara menggunakan library dengan daftar kata kustom.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
console.log('=== Contoh Penggunaan dengan Daftar Kata Kustom ===\n');
|
|
10
|
+
|
|
11
|
+
// ===== Contoh 1: Menggunakan daftar kata kustom =====
|
|
12
|
+
// Daftar kata kotor kustom
|
|
13
|
+
const customBadWords = ['jelek', 'buruk', 'sampah', 'payah', 'lemah', 'amatir'];
|
|
14
|
+
|
|
15
|
+
// Buat instance dengan daftar kata kustom
|
|
16
|
+
const filter = new IDProfanityFilter({
|
|
17
|
+
wordList: customBadWords,
|
|
18
|
+
replaceWith: '#',
|
|
19
|
+
});
|
|
20
|
+
|
|
21
|
+
const teks1 = 'Film itu sangat jelek dan payah, sepertinya dibuat oleh amatir.';
|
|
22
|
+
console.log('Teks asli:', teks1);
|
|
23
|
+
console.log('Hasil filter kustom:', filter.filter(teks1).filtered);
|
|
24
|
+
|
|
25
|
+
// ===== Contoh 2: Menggabungkan daftar kata kustom dengan daftar bawaan =====
|
|
26
|
+
console.log('\n=== Menggabungkan Daftar Kata ===\n');
|
|
27
|
+
|
|
28
|
+
// Reset filter ke pengaturan default
|
|
29
|
+
filter.setOptions({});
|
|
30
|
+
|
|
31
|
+
// Tambahkan kata-kata kustom tapi tetap deteksi kata-kata bawaan
|
|
32
|
+
const teks2 = 'Film itu sangat jelek dan payah, dibuat oleh anjing amatir.';
|
|
33
|
+
console.log('Teks asli:', teks2);
|
|
34
|
+
|
|
35
|
+
// Analisis hanya dengan kata bawaan
|
|
36
|
+
console.log('Analisis standar:');
|
|
37
|
+
const standarAnalisis = filter.analyze(teks2);
|
|
38
|
+
console.log('- Kata terdeteksi:', standarAnalisis.matches);
|
|
39
|
+
|
|
40
|
+
// Tambahkan kata kustom
|
|
41
|
+
filter.setWordList([...customBadWords]);
|
|
42
|
+
console.log('\nAnalisis dengan kata kustom ditambahkan:');
|
|
43
|
+
const customAnalisis = filter.analyze(teks2);
|
|
44
|
+
console.log('- Kata terdeteksi:', customAnalisis.matches);
|
|
45
|
+
|
|
46
|
+
// ===== Contoh 3: Whitelist untuk pengecualian kontekstual =====
|
|
47
|
+
console.log('\n=== Penggunaan Whitelist ===\n');
|
|
48
|
+
|
|
49
|
+
// Kita ingin kata "anjing" diperbolehkan jika dalam konteks binatang
|
|
50
|
+
filter.setOptions({});
|
|
51
|
+
filter.addToWhitelist('anjing');
|
|
52
|
+
|
|
53
|
+
const teks3a = 'Anjing adalah hewan peliharaan yang setia.';
|
|
54
|
+
const teks3b = 'Dasar anjing kamu, tidak punya perasaan!';
|
|
55
|
+
|
|
56
|
+
console.log('Teks dengan konteks binatang:', teks3a);
|
|
57
|
+
console.log('Terdeteksi sebagai kata kotor?', filter.isProfane(teks3a));
|
|
58
|
+
|
|
59
|
+
// Tambahkan kembali "anjing" ke daftar kata kotor
|
|
60
|
+
filter.removeFromWhitelist('anjing');
|
|
61
|
+
console.log('\nSetelah dihapus dari whitelist:');
|
|
62
|
+
console.log('Terdeteksi sebagai kata kotor?', filter.isProfane(teks3a));
|
|
63
|
+
|
|
64
|
+
// ===== Contoh 4: Membuat preset kustom =====
|
|
65
|
+
console.log('\n=== Membuat Preset Kustom ===\n');
|
|
66
|
+
|
|
67
|
+
// Kita bisa membuat preset filter kustom
|
|
68
|
+
const myCustomPreset: FilterOptions = {
|
|
69
|
+
replaceWith: '●', // Karakter pengganti
|
|
70
|
+
fullWordCensor: false,
|
|
71
|
+
keepFirstAndLast: true,
|
|
72
|
+
detectLeetSpeak: true,
|
|
73
|
+
indonesianVariation: true,
|
|
74
|
+
detectSplit: false,
|
|
75
|
+
useRandomGrawlix: false,
|
|
76
|
+
categories: ['insult', 'profanity'],
|
|
77
|
+
severityThreshold: 0.3,
|
|
78
|
+
whitelist: ['anjing'], // Untuk konteks binatang
|
|
79
|
+
};
|
|
80
|
+
|
|
81
|
+
const teks4 = 'Anjing itu lucu, tidak seperti si goblok dan bego itu.';
|
|
82
|
+
console.log('Teks asli:', teks4);
|
|
83
|
+
|
|
84
|
+
// Gunakan preset kustom
|
|
85
|
+
filter.setOptions(myCustomPreset);
|
|
86
|
+
console.log('Hasil preset kustom:', filter.filter(teks4).filtered);
|
|
87
|
+
|
|
88
|
+
// ===== Contoh 5: Filter berdasarkan kategori atau daerah saja =====
|
|
89
|
+
console.log('\n=== Filter Berdasarkan Kategori atau Daerah ===\n');
|
|
90
|
+
|
|
91
|
+
// Filter hanya kata dari kategori "sexual"
|
|
92
|
+
filter.setOptions({
|
|
93
|
+
categories: ['sexual'],
|
|
94
|
+
wordList: [], // Reset daftar kata kustom
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
const teks5 =
|
|
98
|
+
'Ngewe dan bokep itu kata kotor, bego dan anjing tidak terfilter.';
|
|
99
|
+
console.log('Teks asli:', teks5);
|
|
100
|
+
console.log('Filter hanya kategori sexual:', filter.filter(teks5).filtered);
|
|
101
|
+
|
|
102
|
+
// Filter hanya kata dari daerah "jawa"
|
|
103
|
+
filter.setOptions({
|
|
104
|
+
categories: [],
|
|
105
|
+
regions: ['jawa'],
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
const teks5b = 'Kata jancuk dan cuk dari Jawa, anjing adalah kata umum.';
|
|
109
|
+
console.log('\nTeks asli:', teks5b);
|
|
110
|
+
console.log('Filter hanya daerah jawa:', filter.filter(teks5b).filtered);
|
|
111
|
+
|
|
112
|
+
// ===== Contoh 6: Custom wordList dengan metadata =====
|
|
113
|
+
console.log('\n=== Custom wordList dengan Metadata ===\n');
|
|
114
|
+
|
|
115
|
+
// Ini contoh jika Anda ingin membuat daftar kata lengkap dengan metadata
|
|
116
|
+
const customWordsWithMetadata = [
|
|
117
|
+
{
|
|
118
|
+
word: 'jelek',
|
|
119
|
+
category: 'insult' as any,
|
|
120
|
+
region: 'general' as any,
|
|
121
|
+
severity: 0.4,
|
|
122
|
+
aliases: ['jelex', 'jlk'],
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
word: 'payah',
|
|
126
|
+
category: 'insult' as any,
|
|
127
|
+
region: 'general' as any,
|
|
128
|
+
severity: 0.3,
|
|
129
|
+
aliases: ['pyh', 'payaah'],
|
|
130
|
+
},
|
|
131
|
+
];
|
|
132
|
+
|
|
133
|
+
// Namun untuk filter sederhana, cukup gunakan kata-kata saja
|
|
134
|
+
filter.setOptions({
|
|
135
|
+
wordList: customWordsWithMetadata.map((w) => w.word),
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
const teks6 = 'Performanya sangat jelek dan payah sekali.';
|
|
139
|
+
console.log('Teks asli:', teks6);
|
|
140
|
+
console.log('Filter dengan wordList kustom:', filter.filter(teks6).filtered);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@sideid/id-profanity-filter",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.10.4",
|
|
4
4
|
"description": "Library filter kata kotor dalam Bahasa Indonesia",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"module": "dist/index.esm.js",
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
"scripts": {
|
|
9
9
|
"build": "rollup -c",
|
|
10
10
|
"test": "jest",
|
|
11
|
-
"lint": "eslint src
|
|
11
|
+
"lint": "eslint --ext .ts src/",
|
|
12
12
|
"format": "prettier --write \"src/**/*.ts\""
|
|
13
13
|
},
|
|
14
14
|
"keywords": [
|
package/src/config/options.ts
CHANGED
|
@@ -6,7 +6,7 @@ export const general: ProfanityWord[] = [
|
|
|
6
6
|
category: "profanity",
|
|
7
7
|
region: "general",
|
|
8
8
|
severity: 0.7,
|
|
9
|
-
aliases: ["anjay", "anjir", "anying", "njing", "anj"],
|
|
9
|
+
aliases: ["anjay", "anjir", "anying", "njing", "anj", "anjg", "ajg"],
|
|
10
10
|
description: "Mengacu pada hewan anjing, digunakan sebagai umpatan",
|
|
11
11
|
context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
|
|
12
12
|
},
|
|
@@ -15,7 +15,7 @@ export const general: ProfanityWord[] = [
|
|
|
15
15
|
category: "profanity",
|
|
16
16
|
region: "general",
|
|
17
17
|
severity: 0.6,
|
|
18
|
-
aliases: ["bab1", "b4b1"],
|
|
18
|
+
aliases: ["bab1", "b4b1", "b4bi", "8481", "8ab1", "ba81"],
|
|
19
19
|
description: "Mengacu pada hewan babi, digunakan sebagai umpatan",
|
|
20
20
|
context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
|
|
21
21
|
},
|
|
@@ -104,6 +104,106 @@ export const general: ProfanityWord[] = [
|
|
|
104
104
|
description: "Kata yang mengacu pada orang yang banyak bicara",
|
|
105
105
|
context: "Hinaan untuk menyebut orang yang banyak bicara atau cerewet",
|
|
106
106
|
},
|
|
107
|
+
{
|
|
108
|
+
word: "ngentot",
|
|
109
|
+
category: "sexual",
|
|
110
|
+
region: "general",
|
|
111
|
+
severity: 0.9,
|
|
112
|
+
aliases: ["ngentod", "ntot", "tod"],
|
|
113
|
+
description: "Istilah kasar untuk aktivitas seksual",
|
|
114
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual",
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
word: "sialan",
|
|
118
|
+
category: "insult",
|
|
119
|
+
region: "general",
|
|
120
|
+
severity: 0.5,
|
|
121
|
+
aliases: ["sialn", "sl"],
|
|
122
|
+
description: "Kata yang mengacu pada orang yang membawa sial",
|
|
123
|
+
context: "Hinaan untuk menyebut orang yang dianggap membawa sial",
|
|
124
|
+
},
|
|
125
|
+
{
|
|
126
|
+
word: "pler",
|
|
127
|
+
category: "sexual",
|
|
128
|
+
region: "general",
|
|
129
|
+
severity: 0.9,
|
|
130
|
+
aliases: ["peler", "plr", "biji"],
|
|
131
|
+
description: "Istilah kasar untuk alat kelamin laki-laki",
|
|
132
|
+
context: "Kata vulgar yang merujuk pada alat kelamin laki-laki",
|
|
133
|
+
},
|
|
134
|
+
{
|
|
135
|
+
word: "bokep",
|
|
136
|
+
category: "sexual",
|
|
137
|
+
region: "general",
|
|
138
|
+
severity: 0.7,
|
|
139
|
+
aliases: ["bkp", "bokap"],
|
|
140
|
+
description: "Istilah untuk video atau konten pornografi",
|
|
141
|
+
context: "Kata yang mengacu pada materi pornografi",
|
|
142
|
+
},
|
|
143
|
+
{
|
|
144
|
+
word: "coli",
|
|
145
|
+
category: "sexual",
|
|
146
|
+
region: "general",
|
|
147
|
+
severity: 0.8,
|
|
148
|
+
aliases: ["col", "coly"],
|
|
149
|
+
description: "Istilah untuk masturbasi laki-laki",
|
|
150
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual pribadi",
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
word: "desah",
|
|
154
|
+
category: "sexual",
|
|
155
|
+
region: "general",
|
|
156
|
+
severity: 0.6,
|
|
157
|
+
aliases: ["ds4h", "dsh"],
|
|
158
|
+
description: "Istilah untuk suara yang dibuat selama aktivitas seksual",
|
|
159
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
160
|
+
},
|
|
161
|
+
{
|
|
162
|
+
word: "seks",
|
|
163
|
+
category: "sexual",
|
|
164
|
+
region: "general",
|
|
165
|
+
severity: 0.5,
|
|
166
|
+
aliases: ["sex", "ML"],
|
|
167
|
+
description: "Istilah untuk aktivitas seksual",
|
|
168
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
169
|
+
},
|
|
170
|
+
{
|
|
171
|
+
word: "kondom",
|
|
172
|
+
category: "sexual",
|
|
173
|
+
region: "general",
|
|
174
|
+
severity: 0.5,
|
|
175
|
+
aliases: ["kndm", "kondom", "cd"],
|
|
176
|
+
description: "Alat kontrasepsi",
|
|
177
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
178
|
+
},
|
|
179
|
+
{
|
|
180
|
+
word: "ngewe",
|
|
181
|
+
category: "sexual",
|
|
182
|
+
region: "general",
|
|
183
|
+
severity: 0.9,
|
|
184
|
+
aliases: ["ngew", "we"],
|
|
185
|
+
description: "Istilah kasar untuk aktivitas seksual",
|
|
186
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual",
|
|
187
|
+
},
|
|
188
|
+
{
|
|
189
|
+
word: "puki",
|
|
190
|
+
category: "sexual",
|
|
191
|
+
region: "general",
|
|
192
|
+
severity: 0.9,
|
|
193
|
+
aliases: ["puk", "pukih"],
|
|
194
|
+
description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
|
|
195
|
+
context: "Kata vulgar yang merujuk pada anatomi seksual",
|
|
196
|
+
},
|
|
197
|
+
{
|
|
198
|
+
word: "xxx",
|
|
199
|
+
category: "sexual",
|
|
200
|
+
region: "general",
|
|
201
|
+
severity: 0.6,
|
|
202
|
+
aliases: ["xXx", "triplex"],
|
|
203
|
+
description:
|
|
204
|
+
"Simbol yang sering digunakan untuk menandai konten pornografi",
|
|
205
|
+
context: "Digunakan untuk menandai konten seksual eksplisit",
|
|
206
|
+
},
|
|
107
207
|
];
|
|
108
208
|
|
|
109
209
|
export const generalWords = general.map((item) => item.word);
|
|
@@ -95,6 +95,16 @@ export const jawa: ProfanityWord[] = [
|
|
|
95
95
|
context:
|
|
96
96
|
"Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
|
|
97
97
|
},
|
|
98
|
+
{
|
|
99
|
+
word: "itil",
|
|
100
|
+
category: "sexual",
|
|
101
|
+
region: "jawa",
|
|
102
|
+
severity: 0.9,
|
|
103
|
+
aliases: ["itl", "itul"],
|
|
104
|
+
description:
|
|
105
|
+
"Kata vulgar yang mengacu pada bagian dari alat kelamin perempuan",
|
|
106
|
+
context: "Kata vulgar yang merujuk pada anatomi seksual",
|
|
107
|
+
},
|
|
98
108
|
];
|
|
99
109
|
|
|
100
110
|
export const jawaWords = jawa.map((item) => item.word);
|
|
@@ -1,18 +1,17 @@
|
|
|
1
|
-
import { ProfanityWord } from
|
|
1
|
+
import { ProfanityWord } from '../types';
|
|
2
|
+
import { sexualWords } from './categories/sexual';
|
|
3
|
+
import { insultWords } from './categories/insult';
|
|
4
|
+
// import { profanityWords } from './categories/profanity';
|
|
5
|
+
// import { slurWords } from './categories/slur';
|
|
6
|
+
// import { drugsWords } from './categories/drugs';
|
|
7
|
+
// import { disgustingWords } from './categories/disgusting';
|
|
8
|
+
// import { blasphemyWords } from './categories/blasphemy';
|
|
2
9
|
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
// import { disgusting, disgustingWords } from './categories/disgusting';
|
|
9
|
-
// import { blasphemy, blasphemyWords } from './categories/blasphemy';
|
|
10
|
-
|
|
11
|
-
import { general, generalWords } from "./regions/general";
|
|
12
|
-
import { jawa, jawaWords } from "./regions/jawa";
|
|
13
|
-
import { sunda, sundaWords } from "./regions/sunda";
|
|
14
|
-
import { betawi, betawiWords } from "./regions/betawi";
|
|
15
|
-
import { batak, batakWords } from "./regions/batak";
|
|
10
|
+
import { general, generalWords } from './regions/general';
|
|
11
|
+
import { jawa, jawaWords } from './regions/jawa';
|
|
12
|
+
import { sunda, sundaWords } from './regions/sunda';
|
|
13
|
+
import { betawi, betawiWords } from './regions/betawi';
|
|
14
|
+
import { batak, batakWords } from './regions/batak';
|
|
16
15
|
// import { minang, minangWords } from './regions/minang';
|
|
17
16
|
// import { bali, baliWords } from './regions/bali';
|
|
18
17
|
// import { madura, maduraWords } from './regions/madura';
|
|
@@ -96,6 +95,14 @@ export const severeWords: string[] = wordObjects
|
|
|
96
95
|
* @returns Array dari kata kotor
|
|
97
96
|
*/
|
|
98
97
|
export function getWordsByFilter(category?: string, region?: string): string[] {
|
|
98
|
+
if (category && !region && category in wordCategories) {
|
|
99
|
+
return wordCategories[category as keyof typeof wordCategories];
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
if (region && !category && region in wordRegions) {
|
|
103
|
+
return wordRegions[region as keyof typeof wordRegions];
|
|
104
|
+
}
|
|
105
|
+
|
|
99
106
|
return wordObjects
|
|
100
107
|
.filter((word) => {
|
|
101
108
|
const matchCategory = category ? word.category === category : true;
|
package/src/core/analyzer.ts
CHANGED
|
@@ -1,26 +1,20 @@
|
|
|
1
1
|
import {
|
|
2
2
|
FilterOptions,
|
|
3
3
|
AnalysisResult,
|
|
4
|
-
ProfanityWord,
|
|
5
4
|
ProfanityCategory,
|
|
6
5
|
Region,
|
|
7
|
-
} from
|
|
6
|
+
} from '../types';
|
|
8
7
|
import {
|
|
9
8
|
findProfanity,
|
|
10
9
|
findProfanityWithMetadata,
|
|
11
10
|
findCategories,
|
|
12
11
|
findRegions,
|
|
13
12
|
calculateSeverity,
|
|
14
|
-
} from
|
|
15
|
-
import {
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
getContextAroundIndex,
|
|
20
|
-
} from "../utils/stringUtils";
|
|
21
|
-
import { findPossibleProfanityBySimiliarity } from "../utils/similarityUtils";
|
|
22
|
-
import { createContextRegex } from "../utils/regexUtils";
|
|
23
|
-
import { DEFAULT_OPTIONS } from "../config/options";
|
|
13
|
+
} from './matcher';
|
|
14
|
+
import { splitIntoSentences } from '../utils/stringUtils';
|
|
15
|
+
import { findPossibleProfanityBySimiliarity } from '../utils/similarityUtils';
|
|
16
|
+
import { createContextRegex } from '../utils/regexUtils';
|
|
17
|
+
import { DEFAULT_OPTIONS } from '../config/options';
|
|
24
18
|
|
|
25
19
|
/**
|
|
26
20
|
* Menganalisis teks untuk kata kotor
|
|
@@ -206,9 +200,9 @@ export function analyzeWithContext(
|
|
|
206
200
|
|
|
207
201
|
let match;
|
|
208
202
|
while ((match = regex.exec(text)) !== null) {
|
|
209
|
-
const beforeContext = match[1] ||
|
|
203
|
+
const beforeContext = match[1] || '';
|
|
210
204
|
const wordMatch = match[2];
|
|
211
|
-
const afterContext = match[3] ||
|
|
205
|
+
const afterContext = match[3] || '';
|
|
212
206
|
|
|
213
207
|
result.push({
|
|
214
208
|
word: wordMatch,
|
package/src/core/filter.ts
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
|
-
import { FilterOptions, FilterResult, ProfanityWord } from
|
|
2
|
-
import { findProfanity, findProfanityWithMetadata } from
|
|
3
|
-
import { censorWord, escapeRegExp
|
|
4
|
-
import { createWordRegex } from
|
|
5
|
-
import {
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
1
|
+
import { FilterOptions, FilterResult, ProfanityWord } from '../types';
|
|
2
|
+
import { findProfanity, findProfanityWithMetadata } from './matcher';
|
|
3
|
+
import { censorWord, escapeRegExp } from '../utils/stringUtils';
|
|
4
|
+
import { createWordRegex } from '../utils/regexUtils';
|
|
5
|
+
import { DEFAULT_OPTIONS, makeRandomGrawlixString } from '../config/options';
|
|
6
|
+
|
|
7
|
+
interface FindProfanityFunction {
|
|
8
|
+
(text: string, options?: FilterOptions): string[];
|
|
9
|
+
lastActualMatches?: Map<string, string[]>;
|
|
10
|
+
}
|
|
10
11
|
|
|
11
12
|
/**
|
|
12
13
|
* Menyensor kata kotor dalam teks
|
|
@@ -20,7 +21,7 @@ export function filter(
|
|
|
20
21
|
options: FilterOptions = {},
|
|
21
22
|
): FilterResult {
|
|
22
23
|
const {
|
|
23
|
-
replaceWith =
|
|
24
|
+
replaceWith = '*',
|
|
24
25
|
fullWordCensor = true,
|
|
25
26
|
detectLeetSpeak = true,
|
|
26
27
|
whitelist = [],
|
|
@@ -28,6 +29,11 @@ export function filter(
|
|
|
28
29
|
useRandomGrawlix = false,
|
|
29
30
|
keepFirstAndLast = false,
|
|
30
31
|
indonesianVariation = false,
|
|
32
|
+
detectSplit = false,
|
|
33
|
+
detectSimilarity = false,
|
|
34
|
+
useLevenshtein = false,
|
|
35
|
+
maxLevenshteinDistance = 2,
|
|
36
|
+
similarityThreshold = 0.8,
|
|
31
37
|
} = { ...DEFAULT_OPTIONS, ...options };
|
|
32
38
|
|
|
33
39
|
const matches = findProfanity(text, {
|
|
@@ -36,8 +42,16 @@ export function filter(
|
|
|
36
42
|
whitelist,
|
|
37
43
|
checkSubstring,
|
|
38
44
|
indonesianVariation,
|
|
45
|
+
detectSplit,
|
|
46
|
+
detectSimilarity,
|
|
47
|
+
useLevenshtein,
|
|
48
|
+
maxLevenshteinDistance,
|
|
49
|
+
similarityThreshold,
|
|
39
50
|
});
|
|
40
51
|
|
|
52
|
+
const actualMatches: Map<string, string[]> =
|
|
53
|
+
(findProfanity as FindProfanityFunction).lastActualMatches || new Map();
|
|
54
|
+
|
|
41
55
|
const matchDetails = findProfanityWithMetadata(text, options);
|
|
42
56
|
|
|
43
57
|
if (matches.length === 0) {
|
|
@@ -66,49 +80,176 @@ export function filter(
|
|
|
66
80
|
)),
|
|
67
81
|
);
|
|
68
82
|
|
|
69
|
-
const
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
detectSplit: false,
|
|
74
|
-
indonesianVariation: false,
|
|
75
|
-
});
|
|
83
|
+
const variants = actualMatches.get(word.toLowerCase()) || [];
|
|
84
|
+
variants.push(word);
|
|
85
|
+
|
|
86
|
+
const uniqueVariants = [...new Set(variants)];
|
|
76
87
|
|
|
77
|
-
|
|
78
|
-
|
|
88
|
+
uniqueVariants.forEach((variant) => {
|
|
89
|
+
const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, 'gi');
|
|
79
90
|
|
|
80
|
-
|
|
91
|
+
let match;
|
|
92
|
+
while ((match = regex.exec(filteredText)) !== null) {
|
|
93
|
+
const originalWord = match[0];
|
|
81
94
|
|
|
82
|
-
|
|
83
|
-
const originalWord = match[0];
|
|
95
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
84
96
|
|
|
85
|
-
|
|
97
|
+
let censoredWord;
|
|
98
|
+
if (useRandomGrawlix) {
|
|
99
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
100
|
+
} else {
|
|
101
|
+
censoredWord = censorWord(
|
|
102
|
+
originalWord,
|
|
103
|
+
replaceWith,
|
|
104
|
+
!fullWordCensor && keepFirstAndLast,
|
|
105
|
+
);
|
|
106
|
+
}
|
|
86
107
|
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
108
|
+
replacements.push({
|
|
109
|
+
original: originalWord,
|
|
110
|
+
censored: censoredWord,
|
|
111
|
+
metadata,
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
filteredText = filteredText.replace(
|
|
115
|
+
new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'),
|
|
116
|
+
censoredWord,
|
|
95
117
|
);
|
|
96
118
|
}
|
|
119
|
+
});
|
|
97
120
|
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
121
|
+
if (detectSplit || detectLeetSpeak) {
|
|
122
|
+
if (detectLeetSpeak) {
|
|
123
|
+
const leetRegex = createWordRegex(word, {
|
|
124
|
+
wholeWord: true,
|
|
125
|
+
caseSensitive: false,
|
|
126
|
+
leetSpeak: true,
|
|
127
|
+
detectSplit: false,
|
|
128
|
+
indonesianVariation: false,
|
|
129
|
+
});
|
|
103
130
|
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
131
|
+
let match;
|
|
132
|
+
while ((match = leetRegex.exec(filteredText)) !== null) {
|
|
133
|
+
const originalWord = match[0];
|
|
134
|
+
|
|
135
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
136
|
+
|
|
137
|
+
let censoredWord;
|
|
138
|
+
if (useRandomGrawlix) {
|
|
139
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
140
|
+
} else {
|
|
141
|
+
censoredWord = censorWord(
|
|
142
|
+
originalWord,
|
|
143
|
+
replaceWith,
|
|
144
|
+
!fullWordCensor && keepFirstAndLast,
|
|
145
|
+
);
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
replacements.push({
|
|
149
|
+
original: originalWord,
|
|
150
|
+
censored: censoredWord,
|
|
151
|
+
metadata,
|
|
152
|
+
});
|
|
153
|
+
|
|
154
|
+
filteredText = filteredText.replace(
|
|
155
|
+
new RegExp(escapeRegExp(originalWord), 'g'),
|
|
156
|
+
censoredWord,
|
|
157
|
+
);
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
if (detectSplit) {
|
|
162
|
+
const splitRegex = createWordRegex(word, {
|
|
163
|
+
wholeWord: false,
|
|
164
|
+
caseSensitive: false,
|
|
165
|
+
leetSpeak: false,
|
|
166
|
+
detectSplit: true,
|
|
167
|
+
indonesianVariation: false,
|
|
168
|
+
});
|
|
169
|
+
|
|
170
|
+
let match;
|
|
171
|
+
while ((match = splitRegex.exec(filteredText)) !== null) {
|
|
172
|
+
const originalWord = match[0];
|
|
173
|
+
|
|
174
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
175
|
+
|
|
176
|
+
let censoredWord;
|
|
177
|
+
if (useRandomGrawlix) {
|
|
178
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
179
|
+
} else {
|
|
180
|
+
censoredWord = censorWord(
|
|
181
|
+
originalWord,
|
|
182
|
+
replaceWith,
|
|
183
|
+
!fullWordCensor && keepFirstAndLast,
|
|
184
|
+
);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
replacements.push({
|
|
188
|
+
original: originalWord,
|
|
189
|
+
censored: censoredWord,
|
|
190
|
+
metadata,
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
filteredText = filteredText.replace(
|
|
194
|
+
new RegExp(escapeRegExp(originalWord), 'g'),
|
|
195
|
+
censoredWord,
|
|
196
|
+
);
|
|
197
|
+
}
|
|
198
|
+
}
|
|
109
199
|
}
|
|
110
200
|
});
|
|
111
201
|
|
|
202
|
+
if (detectSimilarity && useLevenshtein) {
|
|
203
|
+
matches.forEach((word) => {
|
|
204
|
+
const metadata = matchDetails.find(
|
|
205
|
+
(m) =>
|
|
206
|
+
m.word.toLowerCase() === word.toLowerCase() ||
|
|
207
|
+
(m.aliases &&
|
|
208
|
+
m.aliases.some(
|
|
209
|
+
(alias) => alias.toLowerCase() === word.toLowerCase(),
|
|
210
|
+
)),
|
|
211
|
+
);
|
|
212
|
+
|
|
213
|
+
const variants = actualMatches.get(word.toLowerCase()) || [];
|
|
214
|
+
|
|
215
|
+
variants.forEach((variant) => {
|
|
216
|
+
const exactVariantRegex = new RegExp(
|
|
217
|
+
`\\b${escapeRegExp(variant)}\\b`,
|
|
218
|
+
'gi',
|
|
219
|
+
);
|
|
220
|
+
|
|
221
|
+
let match;
|
|
222
|
+
while ((match = exactVariantRegex.exec(filteredText)) !== null) {
|
|
223
|
+
const originalWord = match[0];
|
|
224
|
+
|
|
225
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
226
|
+
|
|
227
|
+
let censoredWord;
|
|
228
|
+
if (useRandomGrawlix) {
|
|
229
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
230
|
+
} else {
|
|
231
|
+
censoredWord = censorWord(
|
|
232
|
+
originalWord,
|
|
233
|
+
replaceWith,
|
|
234
|
+
!fullWordCensor && keepFirstAndLast,
|
|
235
|
+
);
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
replacements.push({
|
|
239
|
+
original: originalWord,
|
|
240
|
+
censored: censoredWord,
|
|
241
|
+
metadata,
|
|
242
|
+
});
|
|
243
|
+
|
|
244
|
+
filteredText = filteredText.replace(
|
|
245
|
+
new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, 'g'),
|
|
246
|
+
censoredWord,
|
|
247
|
+
);
|
|
248
|
+
}
|
|
249
|
+
});
|
|
250
|
+
});
|
|
251
|
+
}
|
|
252
|
+
|
|
112
253
|
return {
|
|
113
254
|
filtered: filteredText,
|
|
114
255
|
censored: replacements.length,
|