@sideid/id-profanity-filter 1.9.4 → 1.9.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +230 -9
- package/dist/config/options.d.ts +258 -0
- package/dist/constants/categories/index.d.ts +9 -0
- package/dist/constants/categories/insult.d.ts +1 -1
- package/dist/constants/categories/sexual.d.ts +1 -1
- package/dist/constants/regions/batak.d.ts +1 -1
- package/dist/constants/regions/betawi.d.ts +1 -1
- package/dist/constants/regions/general.d.ts +1 -1
- package/dist/constants/regions/index.d.ts +8 -0
- package/dist/constants/regions/jawa.d.ts +1 -1
- package/dist/constants/regions/sunda.d.ts +1 -1
- package/dist/constants/wordList.d.ts +1 -1
- package/dist/core/analyzer.d.ts +4 -4
- package/dist/core/filter.d.ts +1 -1
- package/dist/core/matcher.d.ts +5 -5
- package/dist/index.d.ts +30 -42
- package/dist/index.esm.js +1070 -432
- package/dist/index.esm.js.map +1 -1
- package/dist/index.js +1101 -429
- package/dist/index.js.map +1 -1
- package/dist/types/index.d.ts +29 -41
- package/dist/utils/regexUtils.d.ts +64 -0
- package/dist/utils/similarityUtils.d.ts +57 -0
- package/dist/utils/stringUtils.d.ts +79 -0
- package/examples/advanced.ts +120 -0
- package/examples/basic.ts +60 -41
- package/examples/custom-list.ts +140 -0
- package/package.json +1 -1
- package/src/constants/categories/index.ts +22 -22
- package/src/constants/regions/index.ts +46 -46
package/dist/index.esm.js
CHANGED
|
@@ -1,458 +1,458 @@
|
|
|
1
1
|
const general = [
|
|
2
2
|
{
|
|
3
|
-
word:
|
|
4
|
-
category:
|
|
5
|
-
region:
|
|
3
|
+
word: "anjing",
|
|
4
|
+
category: "profanity",
|
|
5
|
+
region: "general",
|
|
6
6
|
severity: 0.7,
|
|
7
|
-
aliases: [
|
|
8
|
-
description:
|
|
9
|
-
context:
|
|
7
|
+
aliases: ["anjay", "anjir", "anying", "njing", "anj"],
|
|
8
|
+
description: "Mengacu pada hewan anjing, digunakan sebagai umpatan",
|
|
9
|
+
context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
|
|
10
10
|
},
|
|
11
11
|
{
|
|
12
|
-
word:
|
|
13
|
-
category:
|
|
14
|
-
region:
|
|
12
|
+
word: "babi",
|
|
13
|
+
category: "profanity",
|
|
14
|
+
region: "general",
|
|
15
15
|
severity: 0.6,
|
|
16
|
-
aliases: [
|
|
17
|
-
description:
|
|
18
|
-
context:
|
|
16
|
+
aliases: ["bab1", "b4b1"],
|
|
17
|
+
description: "Mengacu pada hewan babi, digunakan sebagai umpatan",
|
|
18
|
+
context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
|
|
19
19
|
},
|
|
20
20
|
{
|
|
21
|
-
word:
|
|
22
|
-
category:
|
|
23
|
-
region:
|
|
21
|
+
word: "bangsat",
|
|
22
|
+
category: "insult",
|
|
23
|
+
region: "general",
|
|
24
24
|
severity: 0.8,
|
|
25
|
-
aliases: [
|
|
26
|
-
description:
|
|
27
|
-
context:
|
|
25
|
+
aliases: ["bangst", "bngst", "bgst"],
|
|
26
|
+
description: "Secara harfiah berarti kutu busuk, digunakan sebagai umpatan untuk menyebut seseorang yang tidak bermoral",
|
|
27
|
+
context: "Umpatan kasar untuk menyebut orang yang dianggap jahat atau merugikan",
|
|
28
28
|
},
|
|
29
29
|
{
|
|
30
|
-
word:
|
|
31
|
-
category:
|
|
32
|
-
region:
|
|
30
|
+
word: "kontol",
|
|
31
|
+
category: "sexual",
|
|
32
|
+
region: "general",
|
|
33
33
|
severity: 0.9,
|
|
34
|
-
aliases: [
|
|
35
|
-
description:
|
|
36
|
-
context:
|
|
34
|
+
aliases: ["kntl", "k0ntol", "k0nt0l"],
|
|
35
|
+
description: "Kata vulgar yang mengacu pada alat kelamin laki-laki",
|
|
36
|
+
context: "Umpatan kasar atau istilah vulgar untuk alat kelamin",
|
|
37
37
|
},
|
|
38
38
|
{
|
|
39
|
-
word:
|
|
40
|
-
category:
|
|
41
|
-
region:
|
|
39
|
+
word: "memek",
|
|
40
|
+
category: "sexual",
|
|
41
|
+
region: "general",
|
|
42
42
|
severity: 0.9,
|
|
43
|
-
aliases: [
|
|
44
|
-
description:
|
|
45
|
-
context:
|
|
43
|
+
aliases: ["mmk", "memk"],
|
|
44
|
+
description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
|
|
45
|
+
context: "Umpatan kasar atau istilah vulgar untuk alat kelamin",
|
|
46
46
|
},
|
|
47
47
|
{
|
|
48
|
-
word:
|
|
49
|
-
category:
|
|
50
|
-
region:
|
|
48
|
+
word: "bego",
|
|
49
|
+
category: "insult",
|
|
50
|
+
region: "general",
|
|
51
51
|
severity: 0.5,
|
|
52
|
-
aliases: [
|
|
53
|
-
description:
|
|
54
|
-
context:
|
|
52
|
+
aliases: ["bgo", "begok"],
|
|
53
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
54
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
55
55
|
},
|
|
56
56
|
{
|
|
57
|
-
word:
|
|
58
|
-
category:
|
|
59
|
-
region:
|
|
57
|
+
word: "tolol",
|
|
58
|
+
category: "insult",
|
|
59
|
+
region: "general",
|
|
60
60
|
severity: 0.6,
|
|
61
|
-
aliases: [
|
|
62
|
-
description:
|
|
63
|
-
context:
|
|
61
|
+
aliases: ["tll", "tlol"],
|
|
62
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
63
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
64
64
|
},
|
|
65
65
|
{
|
|
66
|
-
word:
|
|
67
|
-
category:
|
|
68
|
-
region:
|
|
66
|
+
word: "bajingan",
|
|
67
|
+
category: "insult",
|
|
68
|
+
region: "general",
|
|
69
69
|
severity: 0.7,
|
|
70
|
-
aliases: [
|
|
71
|
-
description:
|
|
72
|
-
context:
|
|
70
|
+
aliases: ["bajingn", "bjgn"],
|
|
71
|
+
description: "Kata yang mengacu pada orang jahat atau tidak bermoral",
|
|
72
|
+
context: "Hinaan untuk menyebut orang yang dianggap jahat atau tidak bermoral",
|
|
73
73
|
},
|
|
74
74
|
{
|
|
75
|
-
word:
|
|
76
|
-
category:
|
|
77
|
-
region:
|
|
75
|
+
word: "goblok",
|
|
76
|
+
category: "insult",
|
|
77
|
+
region: "general",
|
|
78
78
|
severity: 0.6,
|
|
79
|
-
aliases: [
|
|
80
|
-
description:
|
|
81
|
-
context:
|
|
79
|
+
aliases: ["gblk", "goblk"],
|
|
80
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
81
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
82
82
|
},
|
|
83
83
|
{
|
|
84
|
-
word:
|
|
85
|
-
category:
|
|
86
|
-
region:
|
|
84
|
+
word: "keparat",
|
|
85
|
+
category: "insult",
|
|
86
|
+
region: "general",
|
|
87
87
|
severity: 0.7,
|
|
88
|
-
aliases: [
|
|
89
|
-
description:
|
|
90
|
-
context:
|
|
88
|
+
aliases: ["kprt", "keprat"],
|
|
89
|
+
description: "Kata yang mengacu pada orang jahat atau tidak bermoral",
|
|
90
|
+
context: "Hinaan untuk menyebut orang yang dianggap jahat atau tidak bermoral",
|
|
91
91
|
},
|
|
92
92
|
{
|
|
93
|
-
word:
|
|
94
|
-
category:
|
|
95
|
-
region:
|
|
93
|
+
word: "bacot",
|
|
94
|
+
category: "insult",
|
|
95
|
+
region: "general",
|
|
96
96
|
severity: 0.5,
|
|
97
|
-
aliases: [
|
|
98
|
-
description:
|
|
99
|
-
context:
|
|
97
|
+
aliases: ["bcot", "bacot2"],
|
|
98
|
+
description: "Kata yang mengacu pada orang yang banyak bicara",
|
|
99
|
+
context: "Hinaan untuk menyebut orang yang banyak bicara atau cerewet",
|
|
100
100
|
},
|
|
101
101
|
];
|
|
102
102
|
general.map((item) => item.word);
|
|
103
103
|
|
|
104
104
|
const jawa = [
|
|
105
105
|
{
|
|
106
|
-
word:
|
|
107
|
-
category:
|
|
108
|
-
region:
|
|
106
|
+
word: "asu",
|
|
107
|
+
category: "profanity",
|
|
108
|
+
region: "jawa",
|
|
109
109
|
severity: 0.7,
|
|
110
|
-
aliases: [
|
|
111
|
-
description:
|
|
112
|
-
context:
|
|
110
|
+
aliases: ["asyu", "su"],
|
|
111
|
+
description: "Secara harfiah berarti anjing dalam Bahasa Jawa",
|
|
112
|
+
context: "Umpatan umum dalam Bahasa Jawa untuk menunjukkan kemarahan",
|
|
113
113
|
},
|
|
114
114
|
{
|
|
115
|
-
word:
|
|
116
|
-
category:
|
|
117
|
-
region:
|
|
115
|
+
word: "jancok",
|
|
116
|
+
category: "sexual",
|
|
117
|
+
region: "jawa",
|
|
118
118
|
severity: 0.8,
|
|
119
|
-
aliases: [
|
|
120
|
-
description:
|
|
121
|
-
context:
|
|
119
|
+
aliases: ["jancuk", "jncok", "jancuk", "jncuk", "dancok", "dancuk"],
|
|
120
|
+
description: "Kata umpatan kasar dalam Bahasa Jawa",
|
|
121
|
+
context: "Umpatan kasar yang umum digunakan di Jawa Timur",
|
|
122
122
|
},
|
|
123
123
|
{
|
|
124
|
-
word:
|
|
125
|
-
category:
|
|
126
|
-
region:
|
|
124
|
+
word: "cuk",
|
|
125
|
+
category: "sexual",
|
|
126
|
+
region: "jawa",
|
|
127
127
|
severity: 0.7,
|
|
128
|
-
aliases: [
|
|
129
|
-
description:
|
|
130
|
-
context:
|
|
128
|
+
aliases: ["cok", "cook"],
|
|
129
|
+
description: "Singkatan dari jancok/jancuk",
|
|
130
|
+
context: "Umpatan singkat yang umum digunakan di Jawa Timur",
|
|
131
131
|
},
|
|
132
132
|
{
|
|
133
|
-
word:
|
|
134
|
-
category:
|
|
135
|
-
region:
|
|
133
|
+
word: "diancok",
|
|
134
|
+
category: "sexual",
|
|
135
|
+
region: "jawa",
|
|
136
136
|
severity: 0.8,
|
|
137
|
-
aliases: [
|
|
138
|
-
description:
|
|
139
|
-
context:
|
|
137
|
+
aliases: ["diancuk", "dancok", "ancok"],
|
|
138
|
+
description: "Variasi dari jancok/jancuk",
|
|
139
|
+
context: "Umpatan kasar yang umum digunakan di Jawa Timur",
|
|
140
140
|
},
|
|
141
141
|
{
|
|
142
|
-
word:
|
|
143
|
-
category:
|
|
144
|
-
region:
|
|
142
|
+
word: "matamu",
|
|
143
|
+
category: "insult",
|
|
144
|
+
region: "jawa",
|
|
145
145
|
severity: 0.5,
|
|
146
|
-
aliases: [
|
|
147
|
-
description:
|
|
148
|
-
context:
|
|
146
|
+
aliases: ["mripat mu", "matane"],
|
|
147
|
+
description: "Secara harfiah berarti matamu, digunakan sebagai umpatan ringan",
|
|
148
|
+
context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
|
|
149
149
|
},
|
|
150
150
|
{
|
|
151
|
-
word:
|
|
152
|
-
category:
|
|
153
|
-
region:
|
|
151
|
+
word: "mbokne ancok",
|
|
152
|
+
category: "insult",
|
|
153
|
+
region: "jawa",
|
|
154
154
|
severity: 0.8,
|
|
155
|
-
aliases: [
|
|
156
|
-
description:
|
|
157
|
-
context:
|
|
155
|
+
aliases: ["mbokne", "mbokneancok"],
|
|
156
|
+
description: "Umpatan yang menyinggung ibu seseorang",
|
|
157
|
+
context: "Umpatan kasar yang menyinggung orangtua orang lain",
|
|
158
158
|
},
|
|
159
159
|
{
|
|
160
|
-
word:
|
|
161
|
-
category:
|
|
162
|
-
region:
|
|
160
|
+
word: "pekok",
|
|
161
|
+
category: "insult",
|
|
162
|
+
region: "jawa",
|
|
163
163
|
severity: 0.6,
|
|
164
|
-
aliases: [
|
|
165
|
-
description:
|
|
166
|
-
context:
|
|
164
|
+
aliases: ["pekak", "pekilk"],
|
|
165
|
+
description: "Kata hinaan yang menunjukkan kebodohan",
|
|
166
|
+
context: "Hinaan untuk menyebut orang yang dianggap sangat bodoh",
|
|
167
167
|
},
|
|
168
168
|
{
|
|
169
|
-
word:
|
|
170
|
-
category:
|
|
171
|
-
region:
|
|
169
|
+
word: "sempak",
|
|
170
|
+
category: "disgusting",
|
|
171
|
+
region: "jawa",
|
|
172
172
|
severity: 0.5,
|
|
173
|
-
aliases: [
|
|
174
|
-
description:
|
|
175
|
-
context:
|
|
173
|
+
aliases: ["sempok"],
|
|
174
|
+
description: "Mengacu pada pakaian dalam",
|
|
175
|
+
context: "Umpatan ringan yang dianggap jorok",
|
|
176
176
|
},
|
|
177
177
|
{
|
|
178
|
-
word:
|
|
179
|
-
category:
|
|
180
|
-
region:
|
|
178
|
+
word: "taek",
|
|
179
|
+
category: "disgusting",
|
|
180
|
+
region: "jawa",
|
|
181
181
|
severity: 0.6,
|
|
182
|
-
aliases: [
|
|
183
|
-
description:
|
|
184
|
-
context:
|
|
182
|
+
aliases: ["tai", "tahi", "telek"],
|
|
183
|
+
description: "Secara harfiah berarti kotoran/tinja",
|
|
184
|
+
context: "Umpatan untuk menunjukkan sesuatu yang menjijikkan atau buruk",
|
|
185
185
|
},
|
|
186
186
|
{
|
|
187
|
-
word:
|
|
188
|
-
category:
|
|
189
|
-
region:
|
|
187
|
+
word: "ndhasmu",
|
|
188
|
+
category: "insult",
|
|
189
|
+
region: "jawa",
|
|
190
190
|
severity: 0.5,
|
|
191
|
-
aliases: [
|
|
192
|
-
description:
|
|
193
|
-
context:
|
|
191
|
+
aliases: ["ndas mu", "dhasmu"],
|
|
192
|
+
description: "Secara harfiah berarti kepalamu, digunakan sebagai umpatan ringan",
|
|
193
|
+
context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
|
|
194
194
|
},
|
|
195
195
|
];
|
|
196
196
|
jawa.map((item) => item.word);
|
|
197
197
|
|
|
198
198
|
const sunda = [
|
|
199
199
|
{
|
|
200
|
-
word:
|
|
201
|
-
category:
|
|
202
|
-
region:
|
|
200
|
+
word: "bagong",
|
|
201
|
+
category: "insult",
|
|
202
|
+
region: "sunda",
|
|
203
203
|
severity: 0.5,
|
|
204
|
-
aliases: [
|
|
205
|
-
description:
|
|
206
|
-
context:
|
|
204
|
+
aliases: ["bagog"],
|
|
205
|
+
description: "Secara harfiah berarti babi hutan, digunakan sebagai hinaan",
|
|
206
|
+
context: "Hinaan untuk menyebut orang yang dianggap jorok atau rakus",
|
|
207
207
|
},
|
|
208
208
|
{
|
|
209
|
-
word:
|
|
210
|
-
category:
|
|
211
|
-
region:
|
|
209
|
+
word: "belegug",
|
|
210
|
+
category: "insult",
|
|
211
|
+
region: "sunda",
|
|
212
212
|
severity: 0.5,
|
|
213
|
-
aliases: [
|
|
214
|
-
description:
|
|
215
|
-
context:
|
|
213
|
+
aliases: ["belecuk", "beledog"],
|
|
214
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
215
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
216
216
|
},
|
|
217
217
|
{
|
|
218
|
-
word:
|
|
219
|
-
category:
|
|
220
|
-
region:
|
|
218
|
+
word: "goblog",
|
|
219
|
+
category: "insult",
|
|
220
|
+
region: "sunda",
|
|
221
221
|
severity: 0.6,
|
|
222
|
-
aliases: [
|
|
223
|
-
description:
|
|
224
|
-
context:
|
|
222
|
+
aliases: ["goblok", "golbok"],
|
|
223
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
224
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
225
225
|
},
|
|
226
226
|
];
|
|
227
227
|
sunda.map((item) => item.word);
|
|
228
228
|
|
|
229
229
|
const betawi = [
|
|
230
230
|
{
|
|
231
|
-
word:
|
|
232
|
-
category:
|
|
233
|
-
region:
|
|
231
|
+
word: "jablay",
|
|
232
|
+
category: "sexual",
|
|
233
|
+
region: "betawi",
|
|
234
234
|
severity: 0.7,
|
|
235
|
-
aliases: [
|
|
235
|
+
aliases: ["jabl4y", "jalay"],
|
|
236
236
|
description: 'Singkatan dari "jarang dibelai", istilah untuk perempuan yang mudah didekati',
|
|
237
|
-
context:
|
|
237
|
+
context: "Hinaan yang merendahkan untuk wanita",
|
|
238
238
|
},
|
|
239
239
|
{
|
|
240
|
-
word:
|
|
241
|
-
category:
|
|
242
|
-
region:
|
|
240
|
+
word: "udik",
|
|
241
|
+
category: "insult",
|
|
242
|
+
region: "betawi",
|
|
243
243
|
severity: 0.5,
|
|
244
|
-
aliases: [
|
|
245
|
-
description:
|
|
246
|
-
context:
|
|
244
|
+
aliases: ["kampungan", "ndeso"],
|
|
245
|
+
description: "Kata yang mengacu pada seseorang yang dianggap ketinggalan zaman atau tidak modern",
|
|
246
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak modern atau kampungan",
|
|
247
247
|
},
|
|
248
248
|
{
|
|
249
|
-
word:
|
|
250
|
-
category:
|
|
251
|
-
region:
|
|
249
|
+
word: "perek",
|
|
250
|
+
category: "sexual",
|
|
251
|
+
region: "betawi",
|
|
252
252
|
severity: 0.7,
|
|
253
|
-
aliases: [
|
|
254
|
-
description:
|
|
255
|
-
context:
|
|
253
|
+
aliases: ["perempuan eksperimen", "prk"],
|
|
254
|
+
description: "Istilah untuk perempuan yang dianggap memiliki moral yang rendah",
|
|
255
|
+
context: "Hinaan yang merendahkan untuk wanita",
|
|
256
256
|
},
|
|
257
257
|
];
|
|
258
258
|
betawi.map((item) => item.word);
|
|
259
259
|
|
|
260
260
|
const sexual = [
|
|
261
|
-
...general.filter((word) => word.category ===
|
|
262
|
-
...jawa.filter((word) => word.category ===
|
|
263
|
-
...sunda.filter((word) => word.category ===
|
|
264
|
-
...betawi.filter((word) => word.category ===
|
|
265
|
-
{
|
|
266
|
-
word:
|
|
267
|
-
category:
|
|
268
|
-
region:
|
|
261
|
+
...general.filter((word) => word.category === "sexual"),
|
|
262
|
+
...jawa.filter((word) => word.category === "sexual"),
|
|
263
|
+
...sunda.filter((word) => word.category === "sexual"),
|
|
264
|
+
...betawi.filter((word) => word.category === "sexual"),
|
|
265
|
+
{
|
|
266
|
+
word: "bokep",
|
|
267
|
+
category: "sexual",
|
|
268
|
+
region: "general",
|
|
269
269
|
severity: 0.7,
|
|
270
|
-
aliases: [
|
|
271
|
-
description:
|
|
272
|
-
context:
|
|
270
|
+
aliases: ["bkp", "bokap"],
|
|
271
|
+
description: "Istilah untuk video atau konten pornografi",
|
|
272
|
+
context: "Kata yang mengacu pada materi pornografi",
|
|
273
273
|
},
|
|
274
274
|
{
|
|
275
|
-
word:
|
|
276
|
-
category:
|
|
277
|
-
region:
|
|
275
|
+
word: "coli",
|
|
276
|
+
category: "sexual",
|
|
277
|
+
region: "general",
|
|
278
278
|
severity: 0.8,
|
|
279
|
-
aliases: [
|
|
280
|
-
description:
|
|
281
|
-
context:
|
|
279
|
+
aliases: ["col", "coly"],
|
|
280
|
+
description: "Istilah untuk masturbasi laki-laki",
|
|
281
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual pribadi",
|
|
282
282
|
},
|
|
283
283
|
{
|
|
284
|
-
word:
|
|
285
|
-
category:
|
|
286
|
-
region:
|
|
284
|
+
word: "desah",
|
|
285
|
+
category: "sexual",
|
|
286
|
+
region: "general",
|
|
287
287
|
severity: 0.6,
|
|
288
|
-
aliases: [
|
|
289
|
-
description:
|
|
290
|
-
context:
|
|
288
|
+
aliases: ["ds4h", "dsh"],
|
|
289
|
+
description: "Istilah untuk suara yang dibuat selama aktivitas seksual",
|
|
290
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
291
291
|
},
|
|
292
292
|
{
|
|
293
|
-
word:
|
|
294
|
-
category:
|
|
295
|
-
region:
|
|
293
|
+
word: "seks",
|
|
294
|
+
category: "sexual",
|
|
295
|
+
region: "general",
|
|
296
296
|
severity: 0.5,
|
|
297
|
-
aliases: [
|
|
298
|
-
description:
|
|
299
|
-
context:
|
|
297
|
+
aliases: ["sex", "ML"],
|
|
298
|
+
description: "Istilah untuk aktivitas seksual",
|
|
299
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
300
300
|
},
|
|
301
301
|
{
|
|
302
|
-
word:
|
|
303
|
-
category:
|
|
304
|
-
region:
|
|
302
|
+
word: "itil",
|
|
303
|
+
category: "sexual",
|
|
304
|
+
region: "general",
|
|
305
305
|
severity: 0.9,
|
|
306
|
-
aliases: [
|
|
307
|
-
description:
|
|
308
|
-
context:
|
|
306
|
+
aliases: ["itl", "itul"],
|
|
307
|
+
description: "Kata vulgar yang mengacu pada bagian dari alat kelamin perempuan",
|
|
308
|
+
context: "Kata vulgar yang merujuk pada anatomi seksual",
|
|
309
309
|
},
|
|
310
310
|
{
|
|
311
|
-
word:
|
|
312
|
-
category:
|
|
313
|
-
region:
|
|
311
|
+
word: "kondom",
|
|
312
|
+
category: "sexual",
|
|
313
|
+
region: "general",
|
|
314
314
|
severity: 0.5,
|
|
315
|
-
aliases: [
|
|
316
|
-
description:
|
|
317
|
-
context:
|
|
315
|
+
aliases: ["kndm", "kondom", "cd"],
|
|
316
|
+
description: "Alat kontrasepsi",
|
|
317
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
318
318
|
},
|
|
319
319
|
{
|
|
320
|
-
word:
|
|
321
|
-
category:
|
|
322
|
-
region:
|
|
320
|
+
word: "ngewe",
|
|
321
|
+
category: "sexual",
|
|
322
|
+
region: "general",
|
|
323
323
|
severity: 0.9,
|
|
324
|
-
aliases: [
|
|
325
|
-
description:
|
|
326
|
-
context:
|
|
324
|
+
aliases: ["ngew", "we"],
|
|
325
|
+
description: "Istilah kasar untuk aktivitas seksual",
|
|
326
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual",
|
|
327
327
|
},
|
|
328
328
|
{
|
|
329
|
-
word:
|
|
330
|
-
category:
|
|
331
|
-
region:
|
|
329
|
+
word: "puki",
|
|
330
|
+
category: "sexual",
|
|
331
|
+
region: "general",
|
|
332
332
|
severity: 0.9,
|
|
333
|
-
aliases: [
|
|
334
|
-
description:
|
|
335
|
-
context:
|
|
333
|
+
aliases: ["puk", "pukih"],
|
|
334
|
+
description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
|
|
335
|
+
context: "Kata vulgar yang merujuk pada anatomi seksual",
|
|
336
336
|
},
|
|
337
337
|
{
|
|
338
|
-
word:
|
|
339
|
-
category:
|
|
340
|
-
region:
|
|
338
|
+
word: "xxx",
|
|
339
|
+
category: "sexual",
|
|
340
|
+
region: "general",
|
|
341
341
|
severity: 0.6,
|
|
342
|
-
aliases: [
|
|
343
|
-
description:
|
|
344
|
-
context:
|
|
342
|
+
aliases: ["xXx", "triplex"],
|
|
343
|
+
description: "Simbol yang sering digunakan untuk menandai konten pornografi",
|
|
344
|
+
context: "Digunakan untuk menandai konten seksual eksplisit",
|
|
345
345
|
},
|
|
346
346
|
];
|
|
347
347
|
sexual.map((item) => item.word);
|
|
348
348
|
|
|
349
349
|
const insult = [
|
|
350
|
-
...general.filter((word) => word.category ===
|
|
351
|
-
...jawa.filter((word) => word.category ===
|
|
352
|
-
...sunda.filter((word) => word.category ===
|
|
353
|
-
...betawi.filter((word) => word.category ===
|
|
354
|
-
{
|
|
355
|
-
word:
|
|
356
|
-
category:
|
|
357
|
-
region:
|
|
350
|
+
...general.filter((word) => word.category === "insult"),
|
|
351
|
+
...jawa.filter((word) => word.category === "insult"),
|
|
352
|
+
...sunda.filter((word) => word.category === "insult"),
|
|
353
|
+
...betawi.filter((word) => word.category === "insult"),
|
|
354
|
+
{
|
|
355
|
+
word: "idiot",
|
|
356
|
+
category: "insult",
|
|
357
|
+
region: "general",
|
|
358
358
|
severity: 0.6,
|
|
359
|
-
aliases: [
|
|
360
|
-
description:
|
|
361
|
-
context:
|
|
359
|
+
aliases: ["idi0t", "idot"],
|
|
360
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
361
|
+
context: "Hinaan untuk menyebut orang yang dianggap sangat tidak pintar",
|
|
362
362
|
},
|
|
363
363
|
{
|
|
364
|
-
word:
|
|
365
|
-
category:
|
|
366
|
-
region:
|
|
364
|
+
word: "dungu",
|
|
365
|
+
category: "insult",
|
|
366
|
+
region: "general",
|
|
367
367
|
severity: 0.5,
|
|
368
|
-
aliases: [
|
|
369
|
-
description:
|
|
370
|
-
context:
|
|
368
|
+
aliases: ["dngu", "dongu"],
|
|
369
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
370
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
371
371
|
},
|
|
372
372
|
{
|
|
373
|
-
word:
|
|
374
|
-
category:
|
|
375
|
-
region:
|
|
373
|
+
word: "sinting",
|
|
374
|
+
category: "insult",
|
|
375
|
+
region: "general",
|
|
376
376
|
severity: 0.6,
|
|
377
|
-
aliases: [
|
|
378
|
-
description:
|
|
379
|
-
context:
|
|
377
|
+
aliases: ["sintng", "senteng"],
|
|
378
|
+
description: "Kata yang mengacu pada kegilaan atau ketidakwarasan seseorang",
|
|
379
|
+
context: "Hinaan untuk menyebut orang yang dianggap gila atau tidak waras",
|
|
380
380
|
},
|
|
381
381
|
{
|
|
382
|
-
word:
|
|
383
|
-
category:
|
|
384
|
-
region:
|
|
382
|
+
word: "sarap",
|
|
383
|
+
category: "insult",
|
|
384
|
+
region: "general",
|
|
385
385
|
severity: 0.6,
|
|
386
|
-
aliases: [
|
|
387
|
-
description:
|
|
388
|
-
context:
|
|
386
|
+
aliases: ["saraf", "srap"],
|
|
387
|
+
description: "Kata yang mengacu pada kegilaan atau ketidakwarasan seseorang",
|
|
388
|
+
context: "Hinaan untuk menyebut orang yang dianggap gila atau tidak waras",
|
|
389
389
|
},
|
|
390
390
|
{
|
|
391
|
-
word:
|
|
392
|
-
category:
|
|
393
|
-
region:
|
|
391
|
+
word: "geblek",
|
|
392
|
+
category: "insult",
|
|
393
|
+
region: "general",
|
|
394
394
|
severity: 0.5,
|
|
395
|
-
aliases: [
|
|
396
|
-
description:
|
|
397
|
-
context:
|
|
395
|
+
aliases: ["gblk", "geblk"],
|
|
396
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
397
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
398
398
|
},
|
|
399
399
|
{
|
|
400
|
-
word:
|
|
401
|
-
category:
|
|
402
|
-
region:
|
|
400
|
+
word: "kampungan",
|
|
401
|
+
category: "insult",
|
|
402
|
+
region: "general",
|
|
403
403
|
severity: 0.5,
|
|
404
|
-
aliases: [
|
|
405
|
-
description:
|
|
406
|
-
context:
|
|
404
|
+
aliases: ["kmpngn", "kamphungan"],
|
|
405
|
+
description: "Kata yang mengacu pada ketidaksopanan atau kenaifan seseorang",
|
|
406
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak modern atau primitif",
|
|
407
407
|
},
|
|
408
408
|
{
|
|
409
|
-
word:
|
|
410
|
-
category:
|
|
411
|
-
region:
|
|
409
|
+
word: "udik",
|
|
410
|
+
category: "insult",
|
|
411
|
+
region: "general",
|
|
412
412
|
severity: 0.5,
|
|
413
|
-
aliases: [
|
|
414
|
-
description:
|
|
415
|
-
context:
|
|
413
|
+
aliases: ["udhik", "udek"],
|
|
414
|
+
description: "Kata yang mengacu pada ketidaksopanan atau kenaifan seseorang",
|
|
415
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak modern atau primitif",
|
|
416
416
|
},
|
|
417
417
|
{
|
|
418
|
-
word:
|
|
419
|
-
category:
|
|
420
|
-
region:
|
|
418
|
+
word: "dongo",
|
|
419
|
+
category: "insult",
|
|
420
|
+
region: "general",
|
|
421
421
|
severity: 0.5,
|
|
422
|
-
aliases: [
|
|
423
|
-
description:
|
|
424
|
-
context:
|
|
422
|
+
aliases: ["donggo", "dungu"],
|
|
423
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
424
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
425
425
|
},
|
|
426
426
|
{
|
|
427
|
-
word:
|
|
428
|
-
category:
|
|
429
|
-
region:
|
|
427
|
+
word: "tolol",
|
|
428
|
+
category: "insult",
|
|
429
|
+
region: "general",
|
|
430
430
|
severity: 0.6,
|
|
431
|
-
aliases: [
|
|
432
|
-
description:
|
|
433
|
-
context:
|
|
431
|
+
aliases: ["tol0l", "tollo", "tlol"],
|
|
432
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
433
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
434
434
|
},
|
|
435
435
|
{
|
|
436
|
-
word:
|
|
437
|
-
category:
|
|
438
|
-
region:
|
|
436
|
+
word: "brengsek",
|
|
437
|
+
category: "insult",
|
|
438
|
+
region: "general",
|
|
439
439
|
severity: 0.7,
|
|
440
|
-
aliases: [
|
|
441
|
-
description:
|
|
442
|
-
context:
|
|
440
|
+
aliases: ["brengskek", "brengsik", "brengzek"],
|
|
441
|
+
description: "Kata yang mengacu pada kelakuan buruk atau tidak bermoral",
|
|
442
|
+
context: "Hinaan untuk menyebut orang yang dianggap memiliki kelakuan buruk",
|
|
443
443
|
},
|
|
444
444
|
];
|
|
445
445
|
insult.map((item) => item.word);
|
|
446
446
|
|
|
447
447
|
const batak = [
|
|
448
448
|
{
|
|
449
|
-
word:
|
|
450
|
-
category:
|
|
451
|
-
region:
|
|
449
|
+
word: "sundel",
|
|
450
|
+
category: "sexual",
|
|
451
|
+
region: "batak",
|
|
452
452
|
severity: 0.8,
|
|
453
|
-
aliases: [
|
|
454
|
-
description:
|
|
455
|
-
context:
|
|
453
|
+
aliases: ["sundal"],
|
|
454
|
+
description: "Kata kasar untuk menyebut pekerja seks komersial",
|
|
455
|
+
context: "Hinaan kasar untuk wanita",
|
|
456
456
|
},
|
|
457
457
|
];
|
|
458
458
|
batak.map((item) => item.word);
|
|
@@ -485,10 +485,651 @@ wordObjects
|
|
|
485
485
|
.filter((word) => word.severity >= 0.8)
|
|
486
486
|
.map((item) => item.word);
|
|
487
487
|
|
|
488
|
+
/**
|
|
489
|
+
* Menyensor kata dengan karakter pengganti
|
|
490
|
+
*
|
|
491
|
+
* @param word Kata yang akan disensor
|
|
492
|
+
* @param replaceChar Karakter pengganti (default: '*')
|
|
493
|
+
* @param keepFirstAndLast Apakah harus menyimpan huruf pertama dan terakhir
|
|
494
|
+
* @returns Kata yang telah disensor
|
|
495
|
+
*/
|
|
496
|
+
function censorWord(word, replaceChar = "*", keepFirstAndLast = false) {
|
|
497
|
+
if (word.length <= 2) {
|
|
498
|
+
return replaceChar.repeat(word.length);
|
|
499
|
+
}
|
|
500
|
+
if (keepFirstAndLast) {
|
|
501
|
+
return `${word[0]}${replaceChar.repeat(word.length - 2)}${word[word.length - 1]}`;
|
|
502
|
+
}
|
|
503
|
+
return replaceChar.repeat(word.length);
|
|
504
|
+
}
|
|
505
|
+
/**
|
|
506
|
+
* Menormalisasi string untuk keperluan pembandingan
|
|
507
|
+
*
|
|
508
|
+
* @param text Teks yang akan dinormalisasi
|
|
509
|
+
* @returns Teks yang telah dinormalisasi
|
|
510
|
+
*/
|
|
511
|
+
function normalizeText(text) {
|
|
512
|
+
return text
|
|
513
|
+
.toLowerCase()
|
|
514
|
+
.normalize("NFD") // Normalisasi Unicode
|
|
515
|
+
.replace(/[\u0300-\u036f]/g, "") // Hapus diacritic marks
|
|
516
|
+
.replace(/[^\w\s]/g, "") // Hapus karakter non-alphanumeric
|
|
517
|
+
.trim(); // Hapus whitespace di awal dan akhir
|
|
518
|
+
}
|
|
519
|
+
/**
|
|
520
|
+
* Mendeteksi apakah string berisi sebagian atau keseluruhan kata dalam wordList
|
|
521
|
+
*
|
|
522
|
+
* @param text Teks yang akan diperiksa
|
|
523
|
+
* @param wordList Daftar kata yang dicari
|
|
524
|
+
* @param checkSubstring Apakah harus memeriksa substring
|
|
525
|
+
* @returns Boolean apakah teks mengandung kata-kata dalam wordList
|
|
526
|
+
*/
|
|
527
|
+
function containsAnyWord(text, wordList, checkSubstring = false) {
|
|
528
|
+
const normalizedText = normalizeText(text);
|
|
529
|
+
return wordList.some((word) => {
|
|
530
|
+
const normalizedWord = normalizeText(word);
|
|
531
|
+
return checkSubstring
|
|
532
|
+
? normalizedText.includes(normalizedWord)
|
|
533
|
+
: new RegExp(`\\b${escapeRegExp(normalizedWord)}\\b`, "i").test(normalizedText);
|
|
534
|
+
});
|
|
535
|
+
}
|
|
536
|
+
/**
|
|
537
|
+
* Memerikasa apakah stirng merupakan kode untuk kata kotor
|
|
538
|
+
* (Menangkap kasus seperti disensor dengan titik atau garis: a**ing, b*bi, dll)
|
|
539
|
+
*
|
|
540
|
+
* @param text Teks yang akan diperika
|
|
541
|
+
* @param wordList daftar kata kotor
|
|
542
|
+
* @return Boolean apakah teks mengandung kata kotor
|
|
543
|
+
*/
|
|
544
|
+
function containsEuphemism(text, wordList) {
|
|
545
|
+
return wordList.some((word) => {
|
|
546
|
+
if (word.length <= 2)
|
|
547
|
+
return false;
|
|
548
|
+
const firstChar = word[0];
|
|
549
|
+
const lastChar = word[word.length - 1];
|
|
550
|
+
const pattern = new RegExp(`\\b${escapeRegExp(firstChar)}[*@#\\-_.!?\\s]{${word.length - 2}}${escapeRegExp(lastChar)}\\b`, "i");
|
|
551
|
+
return pattern.test(text);
|
|
552
|
+
});
|
|
553
|
+
}
|
|
554
|
+
/**
|
|
555
|
+
* Mendeteksi upaya menghindari filter dengan pemisahan kata
|
|
556
|
+
* (Tangkap kasus seperti: a n j i n g, b-a-b-i, dll)
|
|
557
|
+
*
|
|
558
|
+
* @param text Teks yang akan diperiksa
|
|
559
|
+
* @param wordList Daftar kata kotor
|
|
560
|
+
* @returns Boolean apakah teks mengandung upaya menghindari filter
|
|
561
|
+
*/
|
|
562
|
+
function detectSplitWords(text, wordList) {
|
|
563
|
+
const compressedText = text.replace(/[\s\-_.!?*]/g, "").toLowerCase();
|
|
564
|
+
return wordList.some((word) => compressedText.includes(normalizeText(word)));
|
|
565
|
+
}
|
|
566
|
+
function escapeRegExp(string) {
|
|
567
|
+
return string.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
568
|
+
}
|
|
569
|
+
/**
|
|
570
|
+
* Mengganti sebagian kata dengan masking
|
|
571
|
+
* (berguna untuk email, nomor telepon, dll)
|
|
572
|
+
*
|
|
573
|
+
* @param text Teks untuk dimasking
|
|
574
|
+
* @param visibleStart Jumlah karakter yang terlihat di awal
|
|
575
|
+
* @param visibleEnd Jumlah karakter yang terlihat di akhir
|
|
576
|
+
* @param maskChar Karakter masking
|
|
577
|
+
* @returns Teks yang telah dimasking
|
|
578
|
+
*/
|
|
579
|
+
function maskText(text, visibleStart = 1, visibleEnd = 1, maskChar = "*") {
|
|
580
|
+
if (!text)
|
|
581
|
+
return "";
|
|
582
|
+
if (text.length <= visibleStart + visibleEnd)
|
|
583
|
+
return text;
|
|
584
|
+
const start = text.substring(0, visibleStart);
|
|
585
|
+
const middle = maskChar.repeat(text.length - visibleStart - visibleEnd);
|
|
586
|
+
const end = text.substring(text.length - visibleEnd);
|
|
587
|
+
return start + middle + end;
|
|
588
|
+
}
|
|
589
|
+
/**
|
|
590
|
+
* menguabh sting menjadi bentuk leet speak
|
|
591
|
+
* (untuk testing filter bypass)
|
|
592
|
+
*
|
|
593
|
+
* @param text Teks yang akan diubah
|
|
594
|
+
* @return Teks yang telah diubah ke leet speak
|
|
595
|
+
*/
|
|
596
|
+
function toLeetSpeak(text) {
|
|
597
|
+
const leetMap = {
|
|
598
|
+
a: ["4", "@"],
|
|
599
|
+
b: ["8", "6"],
|
|
600
|
+
c: ["<", "(", "{"],
|
|
601
|
+
e: ["3"],
|
|
602
|
+
g: ["9"],
|
|
603
|
+
i: ["1", "!"],
|
|
604
|
+
l: ["1", "|"],
|
|
605
|
+
o: ["0"],
|
|
606
|
+
s: ["5", "$"],
|
|
607
|
+
t: ["7", "+"],
|
|
608
|
+
z: ["2"],
|
|
609
|
+
};
|
|
610
|
+
return text
|
|
611
|
+
.split("")
|
|
612
|
+
.map((char) => {
|
|
613
|
+
const lowerChar = char.toLowerCase();
|
|
614
|
+
return leetMap[lowerChar] || char;
|
|
615
|
+
})
|
|
616
|
+
.join("");
|
|
617
|
+
}
|
|
618
|
+
/**
|
|
619
|
+
* memisahkan teks menajdi kelimat
|
|
620
|
+
*
|
|
621
|
+
* @param text Teks yang akan dipisahkan
|
|
622
|
+
* @return Array kalimat yang telah dipisahkan
|
|
623
|
+
*/
|
|
624
|
+
function splitIntoSentences(text) {
|
|
625
|
+
// split berdasarkan titik, seru, taya yagn diikuti spasi atau akhir string
|
|
626
|
+
return text
|
|
627
|
+
.split(/(?<=[.!?])\s+|(?<=[.!?])$/)
|
|
628
|
+
.filter((sentence) => sentence.trim().length > 0);
|
|
629
|
+
}
|
|
630
|
+
/**
|
|
631
|
+
* mengambil kata-kata di sekitar indeks tertentu
|
|
632
|
+
*
|
|
633
|
+
* @param text Teks yang akan diambil
|
|
634
|
+
* @param index Indeks dalam teks
|
|
635
|
+
* @param windowSize jumlah kata di sekitar indeks
|
|
636
|
+
* @return Kata-kata di sekitar indeks
|
|
637
|
+
*/
|
|
638
|
+
function getContextAroundIndex(text, index, windowSize = 5) {
|
|
639
|
+
if (!text || index < 0 || index >= text.length)
|
|
640
|
+
return "";
|
|
641
|
+
const words = text.split(/\s+/);
|
|
642
|
+
let currentPosition = 0;
|
|
643
|
+
let targetWordIndex = -1;
|
|
644
|
+
for (let i = 0; i < words.length; i++) {
|
|
645
|
+
const wordLength = words[i].length;
|
|
646
|
+
if (index >= currentPosition && index < currentPosition + wordLength) {
|
|
647
|
+
targetWordIndex = i;
|
|
648
|
+
break;
|
|
649
|
+
}
|
|
650
|
+
// Tambahkan panjang kata dan spasi
|
|
651
|
+
currentPosition += wordLength + 1;
|
|
652
|
+
}
|
|
653
|
+
if (targetWordIndex === -1)
|
|
654
|
+
return "";
|
|
655
|
+
// Ambil kata-kata di sekitar
|
|
656
|
+
const startIndex = Math.max(0, targetWordIndex - windowSize);
|
|
657
|
+
const endIndex = Math.min(words.length, targetWordIndex + windowSize + 1);
|
|
658
|
+
return words.slice(startIndex, endIndex).join(" ");
|
|
659
|
+
}
|
|
660
|
+
|
|
661
|
+
/**
|
|
662
|
+
* Membuat pola regex untuk mencocokkan kata
|
|
663
|
+
*
|
|
664
|
+
* @param word Kata yang akan dibuat pola regex-nya
|
|
665
|
+
* @param options Opsi untuk pembuatan regex
|
|
666
|
+
* @returns Objek RegExp
|
|
667
|
+
*/
|
|
668
|
+
function createWordRegex(word, options = {}) {
|
|
669
|
+
const { wholeWord = true, caseSensitive = false, leetSpeak = true, detectSplit = false, indonesianVariation = false, } = options;
|
|
670
|
+
// Escape karakter khusus regex
|
|
671
|
+
let pattern = escapeRegExp(word);
|
|
672
|
+
// Tambahkan variasi leet speak jika diminta
|
|
673
|
+
if (leetSpeak) {
|
|
674
|
+
pattern = addLeetSpeakVariations(pattern);
|
|
675
|
+
}
|
|
676
|
+
// Tambahkan variasi ejaan Bahasa Indonesia jika diminta
|
|
677
|
+
if (indonesianVariation) {
|
|
678
|
+
pattern = addIndonesianVariations(pattern);
|
|
679
|
+
}
|
|
680
|
+
// Tambahkan kemungkinan split jika diminta
|
|
681
|
+
if (detectSplit) {
|
|
682
|
+
pattern = addSplitVariations(pattern);
|
|
683
|
+
}
|
|
684
|
+
// Tambahkan boundary untuk whole word jika diminta
|
|
685
|
+
if (wholeWord) {
|
|
686
|
+
pattern = `\\b${pattern}\\b`;
|
|
687
|
+
}
|
|
688
|
+
// Buat regex dengan flag case-insensitive jika diminta
|
|
689
|
+
return new RegExp(pattern, caseSensitive ? "g" : "gi");
|
|
690
|
+
}
|
|
691
|
+
/**
|
|
692
|
+
* Menambahkan variasi leet speak ke pola regex
|
|
693
|
+
*
|
|
694
|
+
* Contoh:
|
|
695
|
+
* - 'a' bisa jadi '4', '@'
|
|
696
|
+
* - 'i' bisa jadi '1', '!'
|
|
697
|
+
*
|
|
698
|
+
* @param pattern Pola regex asli
|
|
699
|
+
* @returns Pola regex dengan variasi leet speak
|
|
700
|
+
*/
|
|
701
|
+
function addLeetSpeakVariations(pattern) {
|
|
702
|
+
const leetMap = {
|
|
703
|
+
a: ["a", "4", "@"],
|
|
704
|
+
b: ["b", "8", "6"],
|
|
705
|
+
c: ["c", "(", "{", "<"],
|
|
706
|
+
e: ["e", "3"],
|
|
707
|
+
g: ["g", "6", "9"],
|
|
708
|
+
i: ["i", "1", "!", "|"],
|
|
709
|
+
l: ["l", "1", "|"],
|
|
710
|
+
o: ["o", "0"],
|
|
711
|
+
s: ["s", "5", "$"],
|
|
712
|
+
t: ["t", "7", "+"],
|
|
713
|
+
z: ["z", "2"],
|
|
714
|
+
};
|
|
715
|
+
// Ganti tiap karakter dengan variasinya dalam grup character class
|
|
716
|
+
return pattern
|
|
717
|
+
.split("")
|
|
718
|
+
.map((char) => {
|
|
719
|
+
const lowerChar = char.toLowerCase();
|
|
720
|
+
const variations = leetMap[lowerChar];
|
|
721
|
+
if (variations && variations.length > 1) {
|
|
722
|
+
return `[${variations.join("")}]`;
|
|
723
|
+
}
|
|
724
|
+
return char;
|
|
725
|
+
})
|
|
726
|
+
.join("");
|
|
727
|
+
}
|
|
728
|
+
/**
|
|
729
|
+
* Menambahkan kemungkinan split/pemisahan antar karakter
|
|
730
|
+
*
|
|
731
|
+
* @param pattern Pola regex asli
|
|
732
|
+
* @returns Pola regex dengan kemungkinan split
|
|
733
|
+
*/
|
|
734
|
+
function addSplitVariations(pattern) {
|
|
735
|
+
// Tambahkan kemungkinan spasi atau karakter penghubung di antara setiap huruf
|
|
736
|
+
return pattern.split("").join("[\\s\\-._*+]?");
|
|
737
|
+
}
|
|
738
|
+
/**
|
|
739
|
+
* Membuat regex untuk mencari kata dengan variasi spasi dan karakter penghubung
|
|
740
|
+
* Berguna untuk mendeteksi upaya menghindari filter dengan menambahkan spasi atau karakter lain
|
|
741
|
+
*
|
|
742
|
+
* @param word Kata yang akan dibuat pola regexnya
|
|
743
|
+
* @returns Objek RegExp
|
|
744
|
+
*/
|
|
745
|
+
function createEvasionRegex(word) {
|
|
746
|
+
// Tambahkan kemungkinan spasi atau karakter penghubung di antara setiap huruf
|
|
747
|
+
const pattern = addSplitVariations(escapeRegExp(word));
|
|
748
|
+
return new RegExp(pattern, "gi");
|
|
749
|
+
}
|
|
750
|
+
/**
|
|
751
|
+
* Menambahkan variasi ejaan Bahasa Indonesia
|
|
752
|
+
*
|
|
753
|
+
* @param pattern Pola regex asli
|
|
754
|
+
* @returns Pola regex dengan variasi ejaan Bahasa Indonesia
|
|
755
|
+
*/
|
|
756
|
+
function addIndonesianVariations(pattern) {
|
|
757
|
+
// Variasi ejaan dalam Bahasa Indonesia
|
|
758
|
+
const variationMap = {
|
|
759
|
+
c: ["c", "k"], // contoh: becok/bekok
|
|
760
|
+
k: ["k", "c", "q"], // contoh: kacau/qacau
|
|
761
|
+
j: ["j", "dj"], // contoh: jualan/djualan (ejaan lama)
|
|
762
|
+
y: ["y", "j"], // contoh: ya/ja
|
|
763
|
+
u: ["u", "oe"], // contoh: untuk/oentoek (ejaan lama)
|
|
764
|
+
f: ["f", "p", "v"], // contoh: kafir/kapir
|
|
765
|
+
z: ["z", "j", "s"], // contoh: zaman/jaman
|
|
766
|
+
x: ["x", "ks"], // contoh: taxi/taksi
|
|
767
|
+
};
|
|
768
|
+
// Ganti tiap karakter dengan variasinya
|
|
769
|
+
return pattern
|
|
770
|
+
.split("")
|
|
771
|
+
.map((char) => {
|
|
772
|
+
const lowerChar = char.toLowerCase();
|
|
773
|
+
const variations = variationMap[lowerChar];
|
|
774
|
+
if (variations && variations.length > 1) {
|
|
775
|
+
return `[${variations.join("")}]`;
|
|
776
|
+
}
|
|
777
|
+
return char;
|
|
778
|
+
})
|
|
779
|
+
.join("");
|
|
780
|
+
}
|
|
781
|
+
/**
|
|
782
|
+
* Membuat regex untuk mencocokkan kata dengan mempertimbangkan variasi ejaan Bahasa Indonesia
|
|
783
|
+
*
|
|
784
|
+
* @param word Kata yang akan dibuat pola regexnya
|
|
785
|
+
* @returns Objek RegExp
|
|
786
|
+
*/
|
|
787
|
+
function createIndonesianVariationRegex(word) {
|
|
788
|
+
const pattern = addIndonesianVariations(escapeRegExp(word));
|
|
789
|
+
return new RegExp(`\\b${pattern}\\b`, "gi");
|
|
790
|
+
}
|
|
791
|
+
/**
|
|
792
|
+
* Membuat regex untuk mencocokkan kata dengan konteks
|
|
793
|
+
*
|
|
794
|
+
* @param word Kata yang akan dibuat pola regexnya
|
|
795
|
+
* @param contextSize Jumlah kata konteks sebelum dan sesudah
|
|
796
|
+
* @returns Objek RegExp
|
|
797
|
+
*/
|
|
798
|
+
function createContextRegex(word, contextSize = 3) {
|
|
799
|
+
const wordPattern = escapeRegExp(word);
|
|
800
|
+
// Membuat pola yang menangkap beberapa kata sebelum dan setelah kata target
|
|
801
|
+
const pattern = `((?:\\S+\\s+){0,${contextSize}})(\\b${wordPattern}\\b)((?:\\s+\\S+){0,${contextSize}})`;
|
|
802
|
+
return new RegExp(pattern, "gi");
|
|
803
|
+
}
|
|
804
|
+
/**
|
|
805
|
+
* Membuat regex untuk mencocokkan variasi penulisan kata
|
|
806
|
+
*
|
|
807
|
+
* @param word Kata dasar
|
|
808
|
+
* @returns Objek RegExp untuk mencocokkan berbagai bentuk kata
|
|
809
|
+
*/
|
|
810
|
+
function createWordFormRegex(word) {
|
|
811
|
+
// Implementasi sederhana untuk mencocokkan berbagai imbuhan
|
|
812
|
+
// Ini bisa dikembangkan lebih lanjut untuk mencocokkan bentukan kata yang lebih kompleks
|
|
813
|
+
const prefixes = ["", "me", "pe", "ber", "di", "ter", "se"];
|
|
814
|
+
const suffixes = ["", "kan", "an", "i", "nya"];
|
|
815
|
+
const patterns = [];
|
|
816
|
+
// Kombinasikan prefix dan suffix
|
|
817
|
+
for (const prefix of prefixes) {
|
|
818
|
+
for (const suffix of suffixes) {
|
|
819
|
+
patterns.push(`\\b${prefix}${escapeRegExp(word)}${suffix}\\b`);
|
|
820
|
+
}
|
|
821
|
+
}
|
|
822
|
+
return new RegExp(patterns.join("|"), "gi");
|
|
823
|
+
}
|
|
824
|
+
|
|
825
|
+
/**
|
|
826
|
+
* Menghitung jarak Levenshtein antara dua string
|
|
827
|
+
* (Jumlah operasi insert, delete, atau replace untuk mengubah string1 menjadi string2)
|
|
828
|
+
*
|
|
829
|
+
* @param str1 String pertama
|
|
830
|
+
* @param str2 String kedua
|
|
831
|
+
* @returns Jarak Levenshtein
|
|
832
|
+
*/
|
|
833
|
+
function levenshteinDistance(str1, str2) {
|
|
834
|
+
const s1 = str1.toLowerCase();
|
|
835
|
+
const s2 = str2.toLowerCase();
|
|
836
|
+
const len1 = s1.length;
|
|
837
|
+
const len2 = s2.length;
|
|
838
|
+
// Inisialisasi matrix
|
|
839
|
+
const matrix = [];
|
|
840
|
+
// Inisialisasi baris pertama
|
|
841
|
+
for (let i = 0; i <= len2; i++) {
|
|
842
|
+
matrix[0] = matrix[0] || [];
|
|
843
|
+
matrix[0][i] = i;
|
|
844
|
+
}
|
|
845
|
+
// Inisialisasi kolom pertama
|
|
846
|
+
for (let i = 0; i <= len1; i++) {
|
|
847
|
+
matrix[i] = matrix[i] || [];
|
|
848
|
+
matrix[i][0] = i;
|
|
849
|
+
}
|
|
850
|
+
// Isi matrix
|
|
851
|
+
for (let i = 1; i <= len1; i++) {
|
|
852
|
+
for (let j = 1; j <= len2; j++) {
|
|
853
|
+
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
|
854
|
+
matrix[i][j] = Math.min(matrix[i - 1][j] + 1, // deletion
|
|
855
|
+
matrix[i][j - 1] + 1, // insertion
|
|
856
|
+
matrix[i - 1][j - 1] + cost);
|
|
857
|
+
}
|
|
858
|
+
}
|
|
859
|
+
return matrix[len1][len2];
|
|
860
|
+
}
|
|
861
|
+
/**
|
|
862
|
+
* Menghitung tingkat kesamaan antara dua string
|
|
863
|
+
*
|
|
864
|
+
* @param str1 String pertama
|
|
865
|
+
* @param str2 String kedua
|
|
866
|
+
* @returns Nilai kesamaan (0-1, di mana 1 berarti identik)
|
|
867
|
+
*/
|
|
868
|
+
function stringSimilarity(str1, str2) {
|
|
869
|
+
if (!str1.length && !str2.length)
|
|
870
|
+
return 1;
|
|
871
|
+
if (!str1.length || !str2.length)
|
|
872
|
+
return 0;
|
|
873
|
+
const distance = levenshteinDistance(str1, str2);
|
|
874
|
+
const maxLength = Math.max(str1.length, str2.length);
|
|
875
|
+
return 1 - distance / maxLength;
|
|
876
|
+
}
|
|
877
|
+
/**
|
|
878
|
+
* Mencari string yang paling mirip dari array
|
|
879
|
+
*
|
|
880
|
+
* @param target String target
|
|
881
|
+
* @param candidates Array string kandidat
|
|
882
|
+
* @param threshold Minimum kesamaan yang diterima (0-1)
|
|
883
|
+
* @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
|
|
884
|
+
*/
|
|
885
|
+
function findMostSimilar(target, candidates, threshold = 0.7) {
|
|
886
|
+
if (!candidates.length)
|
|
887
|
+
return null;
|
|
888
|
+
let maxSimilarity = 0;
|
|
889
|
+
let mostSimilar = null;
|
|
890
|
+
for (const candidate of candidates) {
|
|
891
|
+
const similarity = stringSimilarity(target, candidate);
|
|
892
|
+
if (similarity > maxSimilarity && similarity >= threshold) {
|
|
893
|
+
maxSimilarity = similarity;
|
|
894
|
+
mostSimilar = candidate;
|
|
895
|
+
}
|
|
896
|
+
}
|
|
897
|
+
return mostSimilar;
|
|
898
|
+
}
|
|
899
|
+
/**
|
|
900
|
+
* Cek apakah string mungkin merupakan variasi dari kata kotor
|
|
901
|
+
* menggunakan kesamaan string
|
|
902
|
+
*
|
|
903
|
+
* @param input String yang akan diperiksa
|
|
904
|
+
* @param profanityWords Daftar kata kotor
|
|
905
|
+
* @param threshold Batas minimum kesamaan (default: 0.75)
|
|
906
|
+
* @returns Array [Boolean (apakah variasi), String original (jika ditemukan)]
|
|
907
|
+
*/
|
|
908
|
+
function isPossibleProfanityVariation(input, profanityWords, threshold = 0.75) {
|
|
909
|
+
if (!input || !profanityWords.length)
|
|
910
|
+
return [false, null];
|
|
911
|
+
for (const word of profanityWords) {
|
|
912
|
+
const similarity = stringSimilarity(input, word);
|
|
913
|
+
if (similarity >= threshold) {
|
|
914
|
+
return [true, word];
|
|
915
|
+
}
|
|
916
|
+
}
|
|
917
|
+
return [false, null];
|
|
918
|
+
}
|
|
919
|
+
/**
|
|
920
|
+
* Mengelompokkan kata berdasarkan kesamaan
|
|
921
|
+
*
|
|
922
|
+
* @param words Daftar kata
|
|
923
|
+
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
924
|
+
* @returns Array kluster kata yang mirip
|
|
925
|
+
*/
|
|
926
|
+
function clusterSimilarWords(words, threshold = 0.8) {
|
|
927
|
+
const clusters = [];
|
|
928
|
+
const processed = new Set();
|
|
929
|
+
for (const word of words) {
|
|
930
|
+
if (processed.has(word))
|
|
931
|
+
continue;
|
|
932
|
+
const cluster = [word];
|
|
933
|
+
processed.add(word);
|
|
934
|
+
for (const otherWord of words) {
|
|
935
|
+
if (word === otherWord || processed.has(otherWord))
|
|
936
|
+
continue;
|
|
937
|
+
const similarity = stringSimilarity(word, otherWord);
|
|
938
|
+
if (similarity >= threshold) {
|
|
939
|
+
cluster.push(otherWord);
|
|
940
|
+
processed.add(otherWord);
|
|
941
|
+
}
|
|
942
|
+
}
|
|
943
|
+
clusters.push(cluster);
|
|
944
|
+
}
|
|
945
|
+
return clusters;
|
|
946
|
+
}
|
|
947
|
+
/**
|
|
948
|
+
* Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
|
|
949
|
+
*
|
|
950
|
+
* @param text Teks yang akan diperiksa
|
|
951
|
+
* @param profanityWords Daftar kata kotor
|
|
952
|
+
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
953
|
+
* @returns Array kata yang mungkin merupakan kata kotor
|
|
954
|
+
*/
|
|
955
|
+
function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.8) {
|
|
956
|
+
const result = [];
|
|
957
|
+
// Pisahkan teks menjadi kata-kata
|
|
958
|
+
const words = text.toLowerCase().split(/\s+/);
|
|
959
|
+
for (const word of words) {
|
|
960
|
+
// Lewati kata-kata yang terlalu pendek
|
|
961
|
+
if (word.length < 3)
|
|
962
|
+
continue;
|
|
963
|
+
for (const profanity of profanityWords) {
|
|
964
|
+
const similarity = stringSimilarity(word, profanity);
|
|
965
|
+
if (similarity >= threshold) {
|
|
966
|
+
result.push({
|
|
967
|
+
word,
|
|
968
|
+
original: profanity,
|
|
969
|
+
similarity,
|
|
970
|
+
});
|
|
971
|
+
break;
|
|
972
|
+
}
|
|
973
|
+
}
|
|
974
|
+
}
|
|
975
|
+
return result;
|
|
976
|
+
}
|
|
977
|
+
|
|
978
|
+
const DEFAULT_OPTIONS = {
|
|
979
|
+
replaceWith: "*",
|
|
980
|
+
fullWordCensor: true,
|
|
981
|
+
detectLeetSpeak: true,
|
|
982
|
+
checkSubstring: false,
|
|
983
|
+
whitelist: [],
|
|
984
|
+
severityThreshold: 0,
|
|
985
|
+
};
|
|
986
|
+
const FILTER_PRESETS = {
|
|
987
|
+
strict: {
|
|
988
|
+
...DEFAULT_OPTIONS,
|
|
989
|
+
checkSubstring: true,
|
|
990
|
+
detectLeetSpeak: true,
|
|
991
|
+
severityThreshold: 0,
|
|
992
|
+
},
|
|
993
|
+
moderate: {
|
|
994
|
+
...DEFAULT_OPTIONS,
|
|
995
|
+
severityThreshold: 0.5,
|
|
996
|
+
},
|
|
997
|
+
light: {
|
|
998
|
+
...DEFAULT_OPTIONS,
|
|
999
|
+
severityThreshold: 0.7,
|
|
1000
|
+
categories: ["sexual", "slur", "blasphemy"],
|
|
1001
|
+
},
|
|
1002
|
+
childSafe: {
|
|
1003
|
+
...DEFAULT_OPTIONS,
|
|
1004
|
+
checkSubstring: true,
|
|
1005
|
+
detectLeetSpeak: true,
|
|
1006
|
+
fullWordCensor: true,
|
|
1007
|
+
severityThreshold: 0,
|
|
1008
|
+
},
|
|
1009
|
+
};
|
|
1010
|
+
const CATEGORY_PRESETS = {
|
|
1011
|
+
sexual: {
|
|
1012
|
+
...DEFAULT_OPTIONS,
|
|
1013
|
+
categories: ["sexual"],
|
|
1014
|
+
},
|
|
1015
|
+
insults: {
|
|
1016
|
+
...DEFAULT_OPTIONS,
|
|
1017
|
+
categories: ["insult"],
|
|
1018
|
+
},
|
|
1019
|
+
profanity: {
|
|
1020
|
+
...DEFAULT_OPTIONS,
|
|
1021
|
+
categories: ["profanity"],
|
|
1022
|
+
},
|
|
1023
|
+
};
|
|
1024
|
+
const REGION_PRESETS = {
|
|
1025
|
+
general: {
|
|
1026
|
+
...DEFAULT_OPTIONS,
|
|
1027
|
+
regions: ["general"],
|
|
1028
|
+
},
|
|
1029
|
+
jawa: {
|
|
1030
|
+
...DEFAULT_OPTIONS,
|
|
1031
|
+
regions: ["jawa"],
|
|
1032
|
+
},
|
|
1033
|
+
sunda: {
|
|
1034
|
+
...DEFAULT_OPTIONS,
|
|
1035
|
+
regions: ["sunda"],
|
|
1036
|
+
},
|
|
1037
|
+
betawi: {
|
|
1038
|
+
...DEFAULT_OPTIONS,
|
|
1039
|
+
regions: ["betawi"],
|
|
1040
|
+
},
|
|
1041
|
+
batak: {
|
|
1042
|
+
...DEFAULT_OPTIONS,
|
|
1043
|
+
regions: ["batak"],
|
|
1044
|
+
},
|
|
1045
|
+
};
|
|
1046
|
+
const REPLACEMENT_CHARS = {
|
|
1047
|
+
asterisk: "*",
|
|
1048
|
+
hash: "#",
|
|
1049
|
+
dollar: "$",
|
|
1050
|
+
at: "@",
|
|
1051
|
+
percent: "%",
|
|
1052
|
+
underscore: "_",
|
|
1053
|
+
dash: "-",
|
|
1054
|
+
dot: ".",
|
|
1055
|
+
grawlix: "#@$%&!",
|
|
1056
|
+
};
|
|
1057
|
+
/**
|
|
1058
|
+
* Membuat opsi custom dengan menggabungkan dengan default
|
|
1059
|
+
*
|
|
1060
|
+
* @param options Opsi yang akan digabungkan dengan default
|
|
1061
|
+
* @returns Opsi yang sudah digabungkan
|
|
1062
|
+
*/
|
|
1063
|
+
function createOptions(options = {}) {
|
|
1064
|
+
return {
|
|
1065
|
+
...DEFAULT_OPTIONS,
|
|
1066
|
+
...options,
|
|
1067
|
+
};
|
|
1068
|
+
}
|
|
1069
|
+
/**
|
|
1070
|
+
* Mendapatkan opsi dari preset yang ada
|
|
1071
|
+
*
|
|
1072
|
+
* @param presetName Nama preset filter, kategori, atau region
|
|
1073
|
+
* @param additionalOptions Opsi tambahan untuk mengganti preset
|
|
1074
|
+
* @returns Opsi yang sudah digabungkan
|
|
1075
|
+
*/
|
|
1076
|
+
function getPresetOptions(presetName, additionalOptions = {}) {
|
|
1077
|
+
let presetOptions;
|
|
1078
|
+
if (presetName in FILTER_PRESETS) {
|
|
1079
|
+
presetOptions = FILTER_PRESETS[presetName];
|
|
1080
|
+
}
|
|
1081
|
+
else if (presetName in CATEGORY_PRESETS) {
|
|
1082
|
+
presetOptions =
|
|
1083
|
+
CATEGORY_PRESETS[presetName];
|
|
1084
|
+
}
|
|
1085
|
+
else if (presetName in REGION_PRESETS) {
|
|
1086
|
+
presetOptions = REGION_PRESETS[presetName];
|
|
1087
|
+
}
|
|
1088
|
+
else {
|
|
1089
|
+
presetOptions = FILTER_PRESETS.strict;
|
|
1090
|
+
}
|
|
1091
|
+
return {
|
|
1092
|
+
...presetOptions,
|
|
1093
|
+
...additionalOptions,
|
|
1094
|
+
};
|
|
1095
|
+
}
|
|
1096
|
+
/**
|
|
1097
|
+
* Mendapatkan karakter pengganti
|
|
1098
|
+
*
|
|
1099
|
+
* @param type Tipe karakter pengganti
|
|
1100
|
+
* @returns Karakter pengganti
|
|
1101
|
+
*/
|
|
1102
|
+
function getReplacementChar(type = "asterisk") {
|
|
1103
|
+
return REPLACEMENT_CHARS[type] || REPLACEMENT_CHARS.asterisk;
|
|
1104
|
+
}
|
|
1105
|
+
/**
|
|
1106
|
+
* Membuat karakter pengganti random dari grawlix
|
|
1107
|
+
*
|
|
1108
|
+
* @returns Karakter pengganti random
|
|
1109
|
+
*/
|
|
1110
|
+
function getRandomGrawlix() {
|
|
1111
|
+
const grawlix = REPLACEMENT_CHARS.grawlix;
|
|
1112
|
+
return grawlix[Math.floor(Math.random() * grawlix.length)];
|
|
1113
|
+
}
|
|
1114
|
+
/**
|
|
1115
|
+
* Membuat string pengganti untuk kata menggunakan grawlix random
|
|
1116
|
+
*
|
|
1117
|
+
* @param length Panjang string
|
|
1118
|
+
* @returns String pengganti
|
|
1119
|
+
*/
|
|
1120
|
+
function makeRandomGrawlixString(length) {
|
|
1121
|
+
let result = "";
|
|
1122
|
+
const grawlix = REPLACEMENT_CHARS.grawlix;
|
|
1123
|
+
for (let i = 0; i < length; i++) {
|
|
1124
|
+
result += grawlix[Math.floor(Math.random() * grawlix.length)];
|
|
1125
|
+
}
|
|
1126
|
+
return result;
|
|
1127
|
+
}
|
|
1128
|
+
|
|
488
1129
|
function findProfanity(text, options = {}) {
|
|
489
|
-
const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, } = options;
|
|
1130
|
+
const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, } = { ...DEFAULT_OPTIONS, ...options };
|
|
490
1131
|
const normalizedText = normalizeText(text);
|
|
491
|
-
let wordsToCheck = wordList;
|
|
1132
|
+
let wordsToCheck = wordList.length > 0 ? wordList : [];
|
|
492
1133
|
if (wordsToCheck.length === 0) {
|
|
493
1134
|
if (categories || regions || severityThreshold > 0) {
|
|
494
1135
|
wordsToCheck = wordObjects
|
|
@@ -512,99 +1153,68 @@ function findProfanity(text, options = {}) {
|
|
|
512
1153
|
}
|
|
513
1154
|
const matches = new Set();
|
|
514
1155
|
wordsToCheck.forEach((word) => {
|
|
515
|
-
const
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
1156
|
+
const regex = createWordRegex(word, {
|
|
1157
|
+
wholeWord: !checkSubstring,
|
|
1158
|
+
caseSensitive: false,
|
|
1159
|
+
leetSpeak: false,
|
|
1160
|
+
detectSplit: false,
|
|
1161
|
+
indonesianVariation: false,
|
|
1162
|
+
});
|
|
1163
|
+
while ((regex.exec(normalizedText)) !== null) {
|
|
1164
|
+
matches.add(word.toLowerCase());
|
|
520
1165
|
}
|
|
521
1166
|
});
|
|
522
|
-
// jika detectLeetSpeak diaktifkan, cari variasi leet speak
|
|
523
1167
|
if (detectLeetSpeak) {
|
|
524
1168
|
wordsToCheck.forEach((word) => {
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
1169
|
+
const leetRegex = createWordRegex(word, {
|
|
1170
|
+
wholeWord: !checkSubstring,
|
|
1171
|
+
caseSensitive: false,
|
|
1172
|
+
leetSpeak: true,
|
|
1173
|
+
detectSplit: false,
|
|
1174
|
+
indonesianVariation: false,
|
|
1175
|
+
});
|
|
528
1176
|
while ((leetRegex.exec(text)) !== null) {
|
|
529
1177
|
matches.add(word.toLowerCase());
|
|
530
1178
|
}
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
1179
|
+
});
|
|
1180
|
+
}
|
|
1181
|
+
if (indonesianVariation) {
|
|
1182
|
+
wordsToCheck.forEach((word) => {
|
|
1183
|
+
const variantRegex = createWordRegex(word, {
|
|
1184
|
+
wholeWord: !checkSubstring,
|
|
1185
|
+
caseSensitive: false,
|
|
1186
|
+
leetSpeak: false,
|
|
1187
|
+
detectSplit: false,
|
|
1188
|
+
indonesianVariation: true,
|
|
1189
|
+
});
|
|
1190
|
+
while ((variantRegex.exec(text)) !== null) {
|
|
535
1191
|
matches.add(word.toLowerCase());
|
|
536
1192
|
}
|
|
537
1193
|
});
|
|
538
1194
|
}
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
}
|
|
554
|
-
/**
|
|
555
|
-
* Escape karakter khusus dalam regex
|
|
556
|
-
*
|
|
557
|
-
* @param string String untuk di-escape
|
|
558
|
-
* @returns String yang sudah di-escape
|
|
559
|
-
*/
|
|
560
|
-
function escapeRegExp$2(string) {
|
|
561
|
-
return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
562
|
-
}
|
|
563
|
-
/**
|
|
564
|
-
* Pattern untuk leet speak
|
|
565
|
-
*
|
|
566
|
-
* @param word Kata untuk dibuat pattern leet speak
|
|
567
|
-
* @return Pattern regex untuk leet speak
|
|
568
|
-
*/
|
|
569
|
-
function createLeetSpeakPattern(word) {
|
|
570
|
-
const leetMap = {
|
|
571
|
-
a: ['a', '4', '@'],
|
|
572
|
-
b: ['b', '8', '6'],
|
|
573
|
-
c: ['c', '<', '(', '{'],
|
|
574
|
-
e: ['e', '3'],
|
|
575
|
-
g: ['g', '9'],
|
|
576
|
-
i: ['i', '1', '!'],
|
|
577
|
-
l: ['l', '1', '|'],
|
|
578
|
-
o: ['o', '0'],
|
|
579
|
-
s: ['s', '5', '$'],
|
|
580
|
-
t: ['t', '7', '+'],
|
|
581
|
-
z: ['z', '2'],
|
|
582
|
-
};
|
|
583
|
-
return word
|
|
584
|
-
.split('')
|
|
585
|
-
.map((char) => {
|
|
586
|
-
const lowerChar = char.toLowerCase();
|
|
587
|
-
const replacements = leetMap[lowerChar];
|
|
588
|
-
if (replacements && replacements.length > 0) {
|
|
589
|
-
return `[${replacements.join('')}]`;
|
|
590
|
-
}
|
|
591
|
-
else {
|
|
592
|
-
return escapeRegExp$2(char);
|
|
1195
|
+
if (detectSplit) {
|
|
1196
|
+
if (detectSplitWords(text, wordsToCheck)) {
|
|
1197
|
+
wordsToCheck.forEach((word) => {
|
|
1198
|
+
const splitRegex = createWordRegex(word, {
|
|
1199
|
+
wholeWord: false,
|
|
1200
|
+
caseSensitive: false,
|
|
1201
|
+
leetSpeak: false,
|
|
1202
|
+
detectSplit: true,
|
|
1203
|
+
indonesianVariation: false,
|
|
1204
|
+
});
|
|
1205
|
+
if (splitRegex.test(text)) {
|
|
1206
|
+
matches.add(word.toLowerCase());
|
|
1207
|
+
}
|
|
1208
|
+
});
|
|
593
1209
|
}
|
|
594
|
-
}
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
*/
|
|
603
|
-
function createEvasionPattern(word) {
|
|
604
|
-
return word
|
|
605
|
-
.split('')
|
|
606
|
-
.map((char) => escapeRegExp$2(char))
|
|
607
|
-
.join('[\\s\\-._*+]?');
|
|
1210
|
+
}
|
|
1211
|
+
if (detectSimilarity) {
|
|
1212
|
+
const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
|
|
1213
|
+
possibleProfanity.forEach((item) => {
|
|
1214
|
+
matches.add(item.original.toLowerCase());
|
|
1215
|
+
});
|
|
1216
|
+
}
|
|
1217
|
+
return Array.from(matches);
|
|
608
1218
|
}
|
|
609
1219
|
/**
|
|
610
1220
|
* Mencari kata kotor lengkap dengan metadata
|
|
@@ -630,8 +1240,8 @@ function findProfanityWithMetadata(text, options = {}) {
|
|
|
630
1240
|
/**
|
|
631
1241
|
* Mencari kategory kata kotor yang ada dalam teks
|
|
632
1242
|
*
|
|
633
|
-
* @param
|
|
634
|
-
* @
|
|
1243
|
+
* @param matchDetails Hasil pencarian dari fingProfanityWithMetadata()
|
|
1244
|
+
* @return Array kategori unik
|
|
635
1245
|
*/
|
|
636
1246
|
function findCategories(matchDetails) {
|
|
637
1247
|
const categories = new Set();
|
|
@@ -643,8 +1253,8 @@ function findCategories(matchDetails) {
|
|
|
643
1253
|
/**
|
|
644
1254
|
* Mencari region kata kotor yang ada dalam teks
|
|
645
1255
|
*
|
|
646
|
-
* @param
|
|
647
|
-
* @
|
|
1256
|
+
* @param matchDetails Hasil pencarian dari fingProfanityWithMetadata()
|
|
1257
|
+
* @return Array region unik
|
|
648
1258
|
*/
|
|
649
1259
|
function findRegions(matchDetails) {
|
|
650
1260
|
const regions = new Set();
|
|
@@ -693,12 +1303,13 @@ function calculateSeverity(matchDetails) {
|
|
|
693
1303
|
* @returns FilterResult dengan hasil filter
|
|
694
1304
|
*/
|
|
695
1305
|
function filter(text, options = {}) {
|
|
696
|
-
const { replaceWith =
|
|
1306
|
+
const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, } = { ...DEFAULT_OPTIONS, ...options };
|
|
697
1307
|
const matches = findProfanity(text, {
|
|
698
1308
|
...options,
|
|
699
1309
|
detectLeetSpeak,
|
|
700
1310
|
whitelist,
|
|
701
1311
|
checkSubstring,
|
|
1312
|
+
indonesianVariation,
|
|
702
1313
|
});
|
|
703
1314
|
const matchDetails = findProfanityWithMetadata(text, options);
|
|
704
1315
|
if (matches.length === 0) {
|
|
@@ -714,26 +1325,36 @@ function filter(text, options = {}) {
|
|
|
714
1325
|
const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
|
|
715
1326
|
(m.aliases &&
|
|
716
1327
|
m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
|
|
717
|
-
const regex =
|
|
1328
|
+
const regex = createWordRegex(word, {
|
|
1329
|
+
wholeWord: true,
|
|
1330
|
+
caseSensitive: false,
|
|
1331
|
+
leetSpeak: false,
|
|
1332
|
+
detectSplit: false,
|
|
1333
|
+
indonesianVariation: false,
|
|
1334
|
+
});
|
|
718
1335
|
let match;
|
|
719
|
-
|
|
1336
|
+
const textToSearch = filteredText;
|
|
1337
|
+
regex.lastIndex = 0;
|
|
1338
|
+
while ((match = regex.exec(textToSearch)) !== null) {
|
|
720
1339
|
const originalWord = match[0];
|
|
721
1340
|
if (whitelist.includes(originalWord.toLowerCase()))
|
|
722
1341
|
continue;
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
1342
|
+
let censoredWord;
|
|
1343
|
+
if (useRandomGrawlix) {
|
|
1344
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
1345
|
+
}
|
|
1346
|
+
else {
|
|
1347
|
+
censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
|
|
1348
|
+
}
|
|
726
1349
|
replacements.push({
|
|
727
1350
|
original: originalWord,
|
|
728
1351
|
censored: censoredWord,
|
|
729
1352
|
metadata,
|
|
730
1353
|
});
|
|
731
|
-
const replaceRegex = new RegExp(`\\b${escapeRegExp
|
|
1354
|
+
const replaceRegex = new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g");
|
|
732
1355
|
filteredText = filteredText.replace(replaceRegex, censoredWord);
|
|
733
1356
|
}
|
|
734
1357
|
});
|
|
735
|
-
// if (detectLeetSpeak) {
|
|
736
|
-
// }
|
|
737
1358
|
return {
|
|
738
1359
|
filtered: filteredText,
|
|
739
1360
|
censored: replacements.length,
|
|
@@ -751,28 +1372,6 @@ function isProfane(text, options = {}) {
|
|
|
751
1372
|
const matches = findProfanity(text, options);
|
|
752
1373
|
return matches.length > 0;
|
|
753
1374
|
}
|
|
754
|
-
/**
|
|
755
|
-
* Escape karakter khusus regex
|
|
756
|
-
*
|
|
757
|
-
* @param string String untuk di-escape
|
|
758
|
-
* @returns String yang telah di-escape
|
|
759
|
-
*/
|
|
760
|
-
function escapeRegExp$1(string) {
|
|
761
|
-
return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
762
|
-
}
|
|
763
|
-
/**
|
|
764
|
-
* Menyensor sebagian kata
|
|
765
|
-
* @param word Kata yang akan disensor
|
|
766
|
-
* @param replaceChar Karakter pengganti
|
|
767
|
-
* @returns Kata yang sudah disensor sebagian
|
|
768
|
-
*/
|
|
769
|
-
function censorPartialWord(word, replaceChar) {
|
|
770
|
-
if (word.length <= 2) {
|
|
771
|
-
return replaceChar.repeat(word.length);
|
|
772
|
-
}
|
|
773
|
-
// Simpan huruf pertama dan terakhir, sensor yang lain
|
|
774
|
-
return word[0] + replaceChar.repeat(word.length - 2) + word[word.length - 1];
|
|
775
|
-
}
|
|
776
1375
|
|
|
777
1376
|
/**
|
|
778
1377
|
* Menganalisis teks untuk kata kotor
|
|
@@ -782,7 +1381,8 @@ function censorPartialWord(word, replaceChar) {
|
|
|
782
1381
|
* @return AnalysisResult dengan hasil analisis
|
|
783
1382
|
*/
|
|
784
1383
|
function analyze(text, options = {}) {
|
|
785
|
-
const
|
|
1384
|
+
const mergedOptions = { ...DEFAULT_OPTIONS, ...options };
|
|
1385
|
+
const matches = findProfanity(text, mergedOptions);
|
|
786
1386
|
if (matches.length === 0) {
|
|
787
1387
|
return {
|
|
788
1388
|
hasProfanity: false,
|
|
@@ -793,10 +1393,15 @@ function analyze(text, options = {}) {
|
|
|
793
1393
|
severityScore: 0,
|
|
794
1394
|
};
|
|
795
1395
|
}
|
|
796
|
-
const matchDetails = findProfanityWithMetadata(text,
|
|
1396
|
+
const matchDetails = findProfanityWithMetadata(text, mergedOptions);
|
|
797
1397
|
const categories = findCategories(matchDetails);
|
|
798
1398
|
const regions = findRegions(matchDetails);
|
|
799
1399
|
const severityScore = calculateSeverity(matchDetails);
|
|
1400
|
+
let similarWords = [];
|
|
1401
|
+
if (mergedOptions.detectSimilarity) {
|
|
1402
|
+
const wordList = matchDetails.map((word) => word.word);
|
|
1403
|
+
similarWords = findPossibleProfanityBySimiliarity(text, wordList, mergedOptions.similarityThreshold || 0.8);
|
|
1404
|
+
}
|
|
800
1405
|
return {
|
|
801
1406
|
hasProfanity: true,
|
|
802
1407
|
matches,
|
|
@@ -804,17 +1409,19 @@ function analyze(text, options = {}) {
|
|
|
804
1409
|
categories,
|
|
805
1410
|
regions,
|
|
806
1411
|
severityScore,
|
|
1412
|
+
similarWords: mergedOptions.detectSimilarity ? similarWords : undefined,
|
|
807
1413
|
};
|
|
808
1414
|
}
|
|
809
1415
|
/**
|
|
810
|
-
*
|
|
1416
|
+
* Menganalisis daftar teks dan ringkasan
|
|
811
1417
|
*
|
|
812
|
-
* @param texts Daftar teks
|
|
1418
|
+
* @param texts Daftar teks yang dianalisis
|
|
813
1419
|
* @param options Opsi untuk analisis
|
|
814
1420
|
* @return Objek dengan ringkasan analisis
|
|
815
1421
|
*/
|
|
816
1422
|
function batchAnalyze(texts, options = {}) {
|
|
817
|
-
const
|
|
1423
|
+
const mergedOptions = { ...DEFAULT_OPTIONS, ...options };
|
|
1424
|
+
const results = texts.map((text) => analyze(text, mergedOptions));
|
|
818
1425
|
const profaneTexts = results.filter((result) => result.hasProfanity).length;
|
|
819
1426
|
const totalSeverity = results.reduce((sum, result) => sum + result.severityScore, 0);
|
|
820
1427
|
const averageSeverity = profaneTexts > 0 ? totalSeverity / profaneTexts : 0;
|
|
@@ -861,11 +1468,10 @@ function batchAnalyze(texts, options = {}) {
|
|
|
861
1468
|
* @returns Array hasil analisis per-kalimat
|
|
862
1469
|
*/
|
|
863
1470
|
function analyzeBySentence(text, options = {}) {
|
|
864
|
-
const sentences = text
|
|
865
|
-
|
|
866
|
-
.filter((sentence) => sentence.trim().length > 0);
|
|
1471
|
+
const sentences = splitIntoSentences(text);
|
|
1472
|
+
const mergedOptions = { ...DEFAULT_OPTIONS, ...options };
|
|
867
1473
|
return sentences.map((sentence) => {
|
|
868
|
-
const result = analyze(sentence,
|
|
1474
|
+
const result = analyze(sentence, mergedOptions);
|
|
869
1475
|
return {
|
|
870
1476
|
...result,
|
|
871
1477
|
sentence,
|
|
@@ -876,23 +1482,24 @@ function analyzeBySentence(text, options = {}) {
|
|
|
876
1482
|
* Menganalisis teks untuk menemukan kata kotor pada konteks tertentu
|
|
877
1483
|
*
|
|
878
1484
|
* @param text Teks yang akan dianalisis
|
|
879
|
-
* @param
|
|
1485
|
+
* @param contextWindowSize Ukuran konteks (jumlah kata) di sekitar kata kotor
|
|
880
1486
|
* @param options Opsi untuk analisis
|
|
881
1487
|
* @returns Konteks di dekat kata kotor
|
|
882
1488
|
*/
|
|
883
1489
|
function analyzeWithContext(text, contextWindowSize = 5, options = {}) {
|
|
884
|
-
const
|
|
1490
|
+
const mergedOptions = { ...DEFAULT_OPTIONS, ...options };
|
|
1491
|
+
const matches = findProfanity(text, mergedOptions);
|
|
885
1492
|
if (matches.length === 0) {
|
|
886
1493
|
return [];
|
|
887
1494
|
}
|
|
888
1495
|
const result = [];
|
|
889
1496
|
for (const word of matches) {
|
|
890
|
-
const regex =
|
|
1497
|
+
const regex = createContextRegex(word, contextWindowSize);
|
|
891
1498
|
let match;
|
|
892
1499
|
while ((match = regex.exec(text)) !== null) {
|
|
893
|
-
const beforeContext = match[1] ||
|
|
1500
|
+
const beforeContext = match[1] || "";
|
|
894
1501
|
const wordMatch = match[2];
|
|
895
|
-
const afterContext = match[3] ||
|
|
1502
|
+
const afterContext = match[3] || "";
|
|
896
1503
|
result.push({
|
|
897
1504
|
word: wordMatch,
|
|
898
1505
|
context: beforeContext + wordMatch + afterContext,
|
|
@@ -905,9 +1512,6 @@ function analyzeWithContext(text, contextWindowSize = 5, options = {}) {
|
|
|
905
1512
|
}
|
|
906
1513
|
return result;
|
|
907
1514
|
}
|
|
908
|
-
function escapeRegExp(string) {
|
|
909
|
-
return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
910
|
-
}
|
|
911
1515
|
|
|
912
1516
|
class IDProfanityFilter {
|
|
913
1517
|
/**
|
|
@@ -915,7 +1519,7 @@ class IDProfanityFilter {
|
|
|
915
1519
|
* @param options Opsi untuk filter
|
|
916
1520
|
*/
|
|
917
1521
|
constructor(options = {}) {
|
|
918
|
-
this.options = options;
|
|
1522
|
+
this.options = { ...DEFAULT_OPTIONS, ...options };
|
|
919
1523
|
}
|
|
920
1524
|
/**
|
|
921
1525
|
* Menyensor teks yang diberikan
|
|
@@ -976,6 +1580,14 @@ class IDProfanityFilter {
|
|
|
976
1580
|
...options,
|
|
977
1581
|
};
|
|
978
1582
|
}
|
|
1583
|
+
/**
|
|
1584
|
+
* Menggunakan preset yang telah ditentukan
|
|
1585
|
+
* @param presetName Nama preset yang akan digunakan
|
|
1586
|
+
* @param additionalOptions Opsi tambahan untuk override
|
|
1587
|
+
*/
|
|
1588
|
+
usePreset(presetName, additionalOptions = {}) {
|
|
1589
|
+
this.options = getPresetOptions(presetName, additionalOptions);
|
|
1590
|
+
}
|
|
979
1591
|
/**
|
|
980
1592
|
* Menetapkan daftar kata kustom
|
|
981
1593
|
* @param wordList Daftar kata untuk digunakan
|
|
@@ -1002,13 +1614,39 @@ class IDProfanityFilter {
|
|
|
1002
1614
|
return;
|
|
1003
1615
|
this.options.whitelist = this.options.whitelist.filter((w) => w.toLowerCase() !== word.toLowerCase());
|
|
1004
1616
|
}
|
|
1617
|
+
/**
|
|
1618
|
+
* Mengaktifkan deteksi variasi ejaan Indonesia
|
|
1619
|
+
*/
|
|
1620
|
+
enableIndonesianVariations() {
|
|
1621
|
+
this.options.indonesianVariation = true;
|
|
1622
|
+
}
|
|
1623
|
+
/**
|
|
1624
|
+
* Mengaktifkan deteksi kata yang dipisah
|
|
1625
|
+
*/
|
|
1626
|
+
enableSplitWordDetection() {
|
|
1627
|
+
this.options.detectSplit = true;
|
|
1628
|
+
}
|
|
1629
|
+
/**
|
|
1630
|
+
* Mengaktifkan deteksi berdasarkan kesamaan
|
|
1631
|
+
* @param threshold Threshold kesamaan (0-1)
|
|
1632
|
+
*/
|
|
1633
|
+
enableSimilarityDetection(threshold = 0.8) {
|
|
1634
|
+
this.options.detectSimilarity = true;
|
|
1635
|
+
this.options.similarityThreshold = threshold;
|
|
1636
|
+
}
|
|
1005
1637
|
}
|
|
1006
1638
|
const idFilter = {
|
|
1007
|
-
filter: (text, options) => filter(text, options),
|
|
1008
|
-
isProfane: (text, options) => isProfane(text, options),
|
|
1009
|
-
analyze: (text, options) => analyze(text, options),
|
|
1010
|
-
batchAnalyze: (texts, options) => batchAnalyze(texts, options),
|
|
1639
|
+
filter: (text, options) => filter(text, { ...DEFAULT_OPTIONS, ...options }),
|
|
1640
|
+
isProfane: (text, options) => isProfane(text, { ...DEFAULT_OPTIONS, ...options }),
|
|
1641
|
+
analyze: (text, options) => analyze(text, { ...DEFAULT_OPTIONS, ...options }),
|
|
1642
|
+
batchAnalyze: (texts, options) => batchAnalyze(texts, { ...DEFAULT_OPTIONS, ...options }),
|
|
1643
|
+
getPresetOptions,
|
|
1644
|
+
presets: {
|
|
1645
|
+
filter: FILTER_PRESETS,
|
|
1646
|
+
category: CATEGORY_PRESETS,
|
|
1647
|
+
region: REGION_PRESETS,
|
|
1648
|
+
},
|
|
1011
1649
|
};
|
|
1012
1650
|
|
|
1013
|
-
export { IDProfanityFilter, analyze, analyzeBySentence, analyzeWithContext, batchAnalyze, calculateSeverity, IDProfanityFilter as default, filter, findCategories, findProfanity, findProfanityWithMetadata, findRegions, idFilter, isProfane };
|
|
1651
|
+
export { CATEGORY_PRESETS, DEFAULT_OPTIONS, FILTER_PRESETS, IDProfanityFilter, REGION_PRESETS, REPLACEMENT_CHARS, addIndonesianVariations, addLeetSpeakVariations, addSplitVariations, analyze, analyzeBySentence, analyzeWithContext, batchAnalyze, calculateSeverity, censorWord, clusterSimilarWords, containsAnyWord, containsEuphemism, createContextRegex, createEvasionRegex, createIndonesianVariationRegex, createOptions, createWordFormRegex, createWordRegex, IDProfanityFilter as default, detectSplitWords, escapeRegExp, filter, findCategories, findMostSimilar, findPossibleProfanityBySimiliarity, findProfanity, findProfanityWithMetadata, findRegions, getContextAroundIndex, getPresetOptions, getRandomGrawlix, getReplacementChar, idFilter, isPossibleProfanityVariation, isProfane, levenshteinDistance, makeRandomGrawlixString, maskText, normalizeText, splitIntoSentences, stringSimilarity, toLeetSpeak };
|
|
1014
1652
|
//# sourceMappingURL=index.esm.js.map
|