@sideid/id-profanity-filter 1.9.4 → 1.9.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +230 -9
- package/dist/config/options.d.ts +258 -0
- package/dist/constants/categories/index.d.ts +9 -0
- package/dist/constants/categories/insult.d.ts +1 -1
- package/dist/constants/categories/sexual.d.ts +1 -1
- package/dist/constants/regions/batak.d.ts +1 -1
- package/dist/constants/regions/betawi.d.ts +1 -1
- package/dist/constants/regions/general.d.ts +1 -1
- package/dist/constants/regions/index.d.ts +8 -0
- package/dist/constants/regions/jawa.d.ts +1 -1
- package/dist/constants/regions/sunda.d.ts +1 -1
- package/dist/constants/wordList.d.ts +1 -1
- package/dist/core/analyzer.d.ts +4 -4
- package/dist/core/filter.d.ts +1 -1
- package/dist/core/matcher.d.ts +5 -5
- package/dist/index.d.ts +30 -42
- package/dist/index.esm.js +1070 -432
- package/dist/index.esm.js.map +1 -1
- package/dist/index.js +1101 -429
- package/dist/index.js.map +1 -1
- package/dist/types/index.d.ts +29 -41
- package/dist/utils/regexUtils.d.ts +64 -0
- package/dist/utils/similarityUtils.d.ts +57 -0
- package/dist/utils/stringUtils.d.ts +79 -0
- package/examples/advanced.ts +120 -0
- package/examples/basic.ts +60 -41
- package/examples/custom-list.ts +140 -0
- package/package.json +1 -1
- package/src/constants/categories/index.ts +22 -22
- package/src/constants/regions/index.ts +46 -46
package/dist/index.js
CHANGED
|
@@ -4,459 +4,459 @@ Object.defineProperty(exports, '__esModule', { value: true });
|
|
|
4
4
|
|
|
5
5
|
const general = [
|
|
6
6
|
{
|
|
7
|
-
word:
|
|
8
|
-
category:
|
|
9
|
-
region:
|
|
7
|
+
word: "anjing",
|
|
8
|
+
category: "profanity",
|
|
9
|
+
region: "general",
|
|
10
10
|
severity: 0.7,
|
|
11
|
-
aliases: [
|
|
12
|
-
description:
|
|
13
|
-
context:
|
|
11
|
+
aliases: ["anjay", "anjir", "anying", "njing", "anj"],
|
|
12
|
+
description: "Mengacu pada hewan anjing, digunakan sebagai umpatan",
|
|
13
|
+
context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
|
|
14
14
|
},
|
|
15
15
|
{
|
|
16
|
-
word:
|
|
17
|
-
category:
|
|
18
|
-
region:
|
|
16
|
+
word: "babi",
|
|
17
|
+
category: "profanity",
|
|
18
|
+
region: "general",
|
|
19
19
|
severity: 0.6,
|
|
20
|
-
aliases: [
|
|
21
|
-
description:
|
|
22
|
-
context:
|
|
20
|
+
aliases: ["bab1", "b4b1"],
|
|
21
|
+
description: "Mengacu pada hewan babi, digunakan sebagai umpatan",
|
|
22
|
+
context: "Umpatan umum untuk menunjukkan kemarahan atau ketidaksetujuan",
|
|
23
23
|
},
|
|
24
24
|
{
|
|
25
|
-
word:
|
|
26
|
-
category:
|
|
27
|
-
region:
|
|
25
|
+
word: "bangsat",
|
|
26
|
+
category: "insult",
|
|
27
|
+
region: "general",
|
|
28
28
|
severity: 0.8,
|
|
29
|
-
aliases: [
|
|
30
|
-
description:
|
|
31
|
-
context:
|
|
29
|
+
aliases: ["bangst", "bngst", "bgst"],
|
|
30
|
+
description: "Secara harfiah berarti kutu busuk, digunakan sebagai umpatan untuk menyebut seseorang yang tidak bermoral",
|
|
31
|
+
context: "Umpatan kasar untuk menyebut orang yang dianggap jahat atau merugikan",
|
|
32
32
|
},
|
|
33
33
|
{
|
|
34
|
-
word:
|
|
35
|
-
category:
|
|
36
|
-
region:
|
|
34
|
+
word: "kontol",
|
|
35
|
+
category: "sexual",
|
|
36
|
+
region: "general",
|
|
37
37
|
severity: 0.9,
|
|
38
|
-
aliases: [
|
|
39
|
-
description:
|
|
40
|
-
context:
|
|
38
|
+
aliases: ["kntl", "k0ntol", "k0nt0l"],
|
|
39
|
+
description: "Kata vulgar yang mengacu pada alat kelamin laki-laki",
|
|
40
|
+
context: "Umpatan kasar atau istilah vulgar untuk alat kelamin",
|
|
41
41
|
},
|
|
42
42
|
{
|
|
43
|
-
word:
|
|
44
|
-
category:
|
|
45
|
-
region:
|
|
43
|
+
word: "memek",
|
|
44
|
+
category: "sexual",
|
|
45
|
+
region: "general",
|
|
46
46
|
severity: 0.9,
|
|
47
|
-
aliases: [
|
|
48
|
-
description:
|
|
49
|
-
context:
|
|
47
|
+
aliases: ["mmk", "memk"],
|
|
48
|
+
description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
|
|
49
|
+
context: "Umpatan kasar atau istilah vulgar untuk alat kelamin",
|
|
50
50
|
},
|
|
51
51
|
{
|
|
52
|
-
word:
|
|
53
|
-
category:
|
|
54
|
-
region:
|
|
52
|
+
word: "bego",
|
|
53
|
+
category: "insult",
|
|
54
|
+
region: "general",
|
|
55
55
|
severity: 0.5,
|
|
56
|
-
aliases: [
|
|
57
|
-
description:
|
|
58
|
-
context:
|
|
56
|
+
aliases: ["bgo", "begok"],
|
|
57
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
58
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
59
59
|
},
|
|
60
60
|
{
|
|
61
|
-
word:
|
|
62
|
-
category:
|
|
63
|
-
region:
|
|
61
|
+
word: "tolol",
|
|
62
|
+
category: "insult",
|
|
63
|
+
region: "general",
|
|
64
64
|
severity: 0.6,
|
|
65
|
-
aliases: [
|
|
66
|
-
description:
|
|
67
|
-
context:
|
|
65
|
+
aliases: ["tll", "tlol"],
|
|
66
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
67
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
68
68
|
},
|
|
69
69
|
{
|
|
70
|
-
word:
|
|
71
|
-
category:
|
|
72
|
-
region:
|
|
70
|
+
word: "bajingan",
|
|
71
|
+
category: "insult",
|
|
72
|
+
region: "general",
|
|
73
73
|
severity: 0.7,
|
|
74
|
-
aliases: [
|
|
75
|
-
description:
|
|
76
|
-
context:
|
|
74
|
+
aliases: ["bajingn", "bjgn"],
|
|
75
|
+
description: "Kata yang mengacu pada orang jahat atau tidak bermoral",
|
|
76
|
+
context: "Hinaan untuk menyebut orang yang dianggap jahat atau tidak bermoral",
|
|
77
77
|
},
|
|
78
78
|
{
|
|
79
|
-
word:
|
|
80
|
-
category:
|
|
81
|
-
region:
|
|
79
|
+
word: "goblok",
|
|
80
|
+
category: "insult",
|
|
81
|
+
region: "general",
|
|
82
82
|
severity: 0.6,
|
|
83
|
-
aliases: [
|
|
84
|
-
description:
|
|
85
|
-
context:
|
|
83
|
+
aliases: ["gblk", "goblk"],
|
|
84
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
85
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
86
86
|
},
|
|
87
87
|
{
|
|
88
|
-
word:
|
|
89
|
-
category:
|
|
90
|
-
region:
|
|
88
|
+
word: "keparat",
|
|
89
|
+
category: "insult",
|
|
90
|
+
region: "general",
|
|
91
91
|
severity: 0.7,
|
|
92
|
-
aliases: [
|
|
93
|
-
description:
|
|
94
|
-
context:
|
|
92
|
+
aliases: ["kprt", "keprat"],
|
|
93
|
+
description: "Kata yang mengacu pada orang jahat atau tidak bermoral",
|
|
94
|
+
context: "Hinaan untuk menyebut orang yang dianggap jahat atau tidak bermoral",
|
|
95
95
|
},
|
|
96
96
|
{
|
|
97
|
-
word:
|
|
98
|
-
category:
|
|
99
|
-
region:
|
|
97
|
+
word: "bacot",
|
|
98
|
+
category: "insult",
|
|
99
|
+
region: "general",
|
|
100
100
|
severity: 0.5,
|
|
101
|
-
aliases: [
|
|
102
|
-
description:
|
|
103
|
-
context:
|
|
101
|
+
aliases: ["bcot", "bacot2"],
|
|
102
|
+
description: "Kata yang mengacu pada orang yang banyak bicara",
|
|
103
|
+
context: "Hinaan untuk menyebut orang yang banyak bicara atau cerewet",
|
|
104
104
|
},
|
|
105
105
|
];
|
|
106
106
|
general.map((item) => item.word);
|
|
107
107
|
|
|
108
108
|
const jawa = [
|
|
109
109
|
{
|
|
110
|
-
word:
|
|
111
|
-
category:
|
|
112
|
-
region:
|
|
110
|
+
word: "asu",
|
|
111
|
+
category: "profanity",
|
|
112
|
+
region: "jawa",
|
|
113
113
|
severity: 0.7,
|
|
114
|
-
aliases: [
|
|
115
|
-
description:
|
|
116
|
-
context:
|
|
114
|
+
aliases: ["asyu", "su"],
|
|
115
|
+
description: "Secara harfiah berarti anjing dalam Bahasa Jawa",
|
|
116
|
+
context: "Umpatan umum dalam Bahasa Jawa untuk menunjukkan kemarahan",
|
|
117
117
|
},
|
|
118
118
|
{
|
|
119
|
-
word:
|
|
120
|
-
category:
|
|
121
|
-
region:
|
|
119
|
+
word: "jancok",
|
|
120
|
+
category: "sexual",
|
|
121
|
+
region: "jawa",
|
|
122
122
|
severity: 0.8,
|
|
123
|
-
aliases: [
|
|
124
|
-
description:
|
|
125
|
-
context:
|
|
123
|
+
aliases: ["jancuk", "jncok", "jancuk", "jncuk", "dancok", "dancuk"],
|
|
124
|
+
description: "Kata umpatan kasar dalam Bahasa Jawa",
|
|
125
|
+
context: "Umpatan kasar yang umum digunakan di Jawa Timur",
|
|
126
126
|
},
|
|
127
127
|
{
|
|
128
|
-
word:
|
|
129
|
-
category:
|
|
130
|
-
region:
|
|
128
|
+
word: "cuk",
|
|
129
|
+
category: "sexual",
|
|
130
|
+
region: "jawa",
|
|
131
131
|
severity: 0.7,
|
|
132
|
-
aliases: [
|
|
133
|
-
description:
|
|
134
|
-
context:
|
|
132
|
+
aliases: ["cok", "cook"],
|
|
133
|
+
description: "Singkatan dari jancok/jancuk",
|
|
134
|
+
context: "Umpatan singkat yang umum digunakan di Jawa Timur",
|
|
135
135
|
},
|
|
136
136
|
{
|
|
137
|
-
word:
|
|
138
|
-
category:
|
|
139
|
-
region:
|
|
137
|
+
word: "diancok",
|
|
138
|
+
category: "sexual",
|
|
139
|
+
region: "jawa",
|
|
140
140
|
severity: 0.8,
|
|
141
|
-
aliases: [
|
|
142
|
-
description:
|
|
143
|
-
context:
|
|
141
|
+
aliases: ["diancuk", "dancok", "ancok"],
|
|
142
|
+
description: "Variasi dari jancok/jancuk",
|
|
143
|
+
context: "Umpatan kasar yang umum digunakan di Jawa Timur",
|
|
144
144
|
},
|
|
145
145
|
{
|
|
146
|
-
word:
|
|
147
|
-
category:
|
|
148
|
-
region:
|
|
146
|
+
word: "matamu",
|
|
147
|
+
category: "insult",
|
|
148
|
+
region: "jawa",
|
|
149
149
|
severity: 0.5,
|
|
150
|
-
aliases: [
|
|
151
|
-
description:
|
|
152
|
-
context:
|
|
150
|
+
aliases: ["mripat mu", "matane"],
|
|
151
|
+
description: "Secara harfiah berarti matamu, digunakan sebagai umpatan ringan",
|
|
152
|
+
context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
|
|
153
153
|
},
|
|
154
154
|
{
|
|
155
|
-
word:
|
|
156
|
-
category:
|
|
157
|
-
region:
|
|
155
|
+
word: "mbokne ancok",
|
|
156
|
+
category: "insult",
|
|
157
|
+
region: "jawa",
|
|
158
158
|
severity: 0.8,
|
|
159
|
-
aliases: [
|
|
160
|
-
description:
|
|
161
|
-
context:
|
|
159
|
+
aliases: ["mbokne", "mbokneancok"],
|
|
160
|
+
description: "Umpatan yang menyinggung ibu seseorang",
|
|
161
|
+
context: "Umpatan kasar yang menyinggung orangtua orang lain",
|
|
162
162
|
},
|
|
163
163
|
{
|
|
164
|
-
word:
|
|
165
|
-
category:
|
|
166
|
-
region:
|
|
164
|
+
word: "pekok",
|
|
165
|
+
category: "insult",
|
|
166
|
+
region: "jawa",
|
|
167
167
|
severity: 0.6,
|
|
168
|
-
aliases: [
|
|
169
|
-
description:
|
|
170
|
-
context:
|
|
168
|
+
aliases: ["pekak", "pekilk"],
|
|
169
|
+
description: "Kata hinaan yang menunjukkan kebodohan",
|
|
170
|
+
context: "Hinaan untuk menyebut orang yang dianggap sangat bodoh",
|
|
171
171
|
},
|
|
172
172
|
{
|
|
173
|
-
word:
|
|
174
|
-
category:
|
|
175
|
-
region:
|
|
173
|
+
word: "sempak",
|
|
174
|
+
category: "disgusting",
|
|
175
|
+
region: "jawa",
|
|
176
176
|
severity: 0.5,
|
|
177
|
-
aliases: [
|
|
178
|
-
description:
|
|
179
|
-
context:
|
|
177
|
+
aliases: ["sempok"],
|
|
178
|
+
description: "Mengacu pada pakaian dalam",
|
|
179
|
+
context: "Umpatan ringan yang dianggap jorok",
|
|
180
180
|
},
|
|
181
181
|
{
|
|
182
|
-
word:
|
|
183
|
-
category:
|
|
184
|
-
region:
|
|
182
|
+
word: "taek",
|
|
183
|
+
category: "disgusting",
|
|
184
|
+
region: "jawa",
|
|
185
185
|
severity: 0.6,
|
|
186
|
-
aliases: [
|
|
187
|
-
description:
|
|
188
|
-
context:
|
|
186
|
+
aliases: ["tai", "tahi", "telek"],
|
|
187
|
+
description: "Secara harfiah berarti kotoran/tinja",
|
|
188
|
+
context: "Umpatan untuk menunjukkan sesuatu yang menjijikkan atau buruk",
|
|
189
189
|
},
|
|
190
190
|
{
|
|
191
|
-
word:
|
|
192
|
-
category:
|
|
193
|
-
region:
|
|
191
|
+
word: "ndhasmu",
|
|
192
|
+
category: "insult",
|
|
193
|
+
region: "jawa",
|
|
194
194
|
severity: 0.5,
|
|
195
|
-
aliases: [
|
|
196
|
-
description:
|
|
197
|
-
context:
|
|
195
|
+
aliases: ["ndas mu", "dhasmu"],
|
|
196
|
+
description: "Secara harfiah berarti kepalamu, digunakan sebagai umpatan ringan",
|
|
197
|
+
context: "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
|
|
198
198
|
},
|
|
199
199
|
];
|
|
200
200
|
jawa.map((item) => item.word);
|
|
201
201
|
|
|
202
202
|
const sunda = [
|
|
203
203
|
{
|
|
204
|
-
word:
|
|
205
|
-
category:
|
|
206
|
-
region:
|
|
204
|
+
word: "bagong",
|
|
205
|
+
category: "insult",
|
|
206
|
+
region: "sunda",
|
|
207
207
|
severity: 0.5,
|
|
208
|
-
aliases: [
|
|
209
|
-
description:
|
|
210
|
-
context:
|
|
208
|
+
aliases: ["bagog"],
|
|
209
|
+
description: "Secara harfiah berarti babi hutan, digunakan sebagai hinaan",
|
|
210
|
+
context: "Hinaan untuk menyebut orang yang dianggap jorok atau rakus",
|
|
211
211
|
},
|
|
212
212
|
{
|
|
213
|
-
word:
|
|
214
|
-
category:
|
|
215
|
-
region:
|
|
213
|
+
word: "belegug",
|
|
214
|
+
category: "insult",
|
|
215
|
+
region: "sunda",
|
|
216
216
|
severity: 0.5,
|
|
217
|
-
aliases: [
|
|
218
|
-
description:
|
|
219
|
-
context:
|
|
217
|
+
aliases: ["belecuk", "beledog"],
|
|
218
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
219
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
220
220
|
},
|
|
221
221
|
{
|
|
222
|
-
word:
|
|
223
|
-
category:
|
|
224
|
-
region:
|
|
222
|
+
word: "goblog",
|
|
223
|
+
category: "insult",
|
|
224
|
+
region: "sunda",
|
|
225
225
|
severity: 0.6,
|
|
226
|
-
aliases: [
|
|
227
|
-
description:
|
|
228
|
-
context:
|
|
226
|
+
aliases: ["goblok", "golbok"],
|
|
227
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
228
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
229
229
|
},
|
|
230
230
|
];
|
|
231
231
|
sunda.map((item) => item.word);
|
|
232
232
|
|
|
233
233
|
const betawi = [
|
|
234
234
|
{
|
|
235
|
-
word:
|
|
236
|
-
category:
|
|
237
|
-
region:
|
|
235
|
+
word: "jablay",
|
|
236
|
+
category: "sexual",
|
|
237
|
+
region: "betawi",
|
|
238
238
|
severity: 0.7,
|
|
239
|
-
aliases: [
|
|
239
|
+
aliases: ["jabl4y", "jalay"],
|
|
240
240
|
description: 'Singkatan dari "jarang dibelai", istilah untuk perempuan yang mudah didekati',
|
|
241
|
-
context:
|
|
241
|
+
context: "Hinaan yang merendahkan untuk wanita",
|
|
242
242
|
},
|
|
243
243
|
{
|
|
244
|
-
word:
|
|
245
|
-
category:
|
|
246
|
-
region:
|
|
244
|
+
word: "udik",
|
|
245
|
+
category: "insult",
|
|
246
|
+
region: "betawi",
|
|
247
247
|
severity: 0.5,
|
|
248
|
-
aliases: [
|
|
249
|
-
description:
|
|
250
|
-
context:
|
|
248
|
+
aliases: ["kampungan", "ndeso"],
|
|
249
|
+
description: "Kata yang mengacu pada seseorang yang dianggap ketinggalan zaman atau tidak modern",
|
|
250
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak modern atau kampungan",
|
|
251
251
|
},
|
|
252
252
|
{
|
|
253
|
-
word:
|
|
254
|
-
category:
|
|
255
|
-
region:
|
|
253
|
+
word: "perek",
|
|
254
|
+
category: "sexual",
|
|
255
|
+
region: "betawi",
|
|
256
256
|
severity: 0.7,
|
|
257
|
-
aliases: [
|
|
258
|
-
description:
|
|
259
|
-
context:
|
|
257
|
+
aliases: ["perempuan eksperimen", "prk"],
|
|
258
|
+
description: "Istilah untuk perempuan yang dianggap memiliki moral yang rendah",
|
|
259
|
+
context: "Hinaan yang merendahkan untuk wanita",
|
|
260
260
|
},
|
|
261
261
|
];
|
|
262
262
|
betawi.map((item) => item.word);
|
|
263
263
|
|
|
264
264
|
const sexual = [
|
|
265
|
-
...general.filter((word) => word.category ===
|
|
266
|
-
...jawa.filter((word) => word.category ===
|
|
267
|
-
...sunda.filter((word) => word.category ===
|
|
268
|
-
...betawi.filter((word) => word.category ===
|
|
265
|
+
...general.filter((word) => word.category === "sexual"),
|
|
266
|
+
...jawa.filter((word) => word.category === "sexual"),
|
|
267
|
+
...sunda.filter((word) => word.category === "sexual"),
|
|
268
|
+
...betawi.filter((word) => word.category === "sexual"),
|
|
269
269
|
{
|
|
270
|
-
word:
|
|
271
|
-
category:
|
|
272
|
-
region:
|
|
270
|
+
word: "bokep",
|
|
271
|
+
category: "sexual",
|
|
272
|
+
region: "general",
|
|
273
273
|
severity: 0.7,
|
|
274
|
-
aliases: [
|
|
275
|
-
description:
|
|
276
|
-
context:
|
|
274
|
+
aliases: ["bkp", "bokap"],
|
|
275
|
+
description: "Istilah untuk video atau konten pornografi",
|
|
276
|
+
context: "Kata yang mengacu pada materi pornografi",
|
|
277
277
|
},
|
|
278
278
|
{
|
|
279
|
-
word:
|
|
280
|
-
category:
|
|
281
|
-
region:
|
|
279
|
+
word: "coli",
|
|
280
|
+
category: "sexual",
|
|
281
|
+
region: "general",
|
|
282
282
|
severity: 0.8,
|
|
283
|
-
aliases: [
|
|
284
|
-
description:
|
|
285
|
-
context:
|
|
283
|
+
aliases: ["col", "coly"],
|
|
284
|
+
description: "Istilah untuk masturbasi laki-laki",
|
|
285
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual pribadi",
|
|
286
286
|
},
|
|
287
287
|
{
|
|
288
|
-
word:
|
|
289
|
-
category:
|
|
290
|
-
region:
|
|
288
|
+
word: "desah",
|
|
289
|
+
category: "sexual",
|
|
290
|
+
region: "general",
|
|
291
291
|
severity: 0.6,
|
|
292
|
-
aliases: [
|
|
293
|
-
description:
|
|
294
|
-
context:
|
|
292
|
+
aliases: ["ds4h", "dsh"],
|
|
293
|
+
description: "Istilah untuk suara yang dibuat selama aktivitas seksual",
|
|
294
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
295
295
|
},
|
|
296
296
|
{
|
|
297
|
-
word:
|
|
298
|
-
category:
|
|
299
|
-
region:
|
|
297
|
+
word: "seks",
|
|
298
|
+
category: "sexual",
|
|
299
|
+
region: "general",
|
|
300
300
|
severity: 0.5,
|
|
301
|
-
aliases: [
|
|
302
|
-
description:
|
|
303
|
-
context:
|
|
301
|
+
aliases: ["sex", "ML"],
|
|
302
|
+
description: "Istilah untuk aktivitas seksual",
|
|
303
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
304
304
|
},
|
|
305
305
|
{
|
|
306
|
-
word:
|
|
307
|
-
category:
|
|
308
|
-
region:
|
|
306
|
+
word: "itil",
|
|
307
|
+
category: "sexual",
|
|
308
|
+
region: "general",
|
|
309
309
|
severity: 0.9,
|
|
310
|
-
aliases: [
|
|
311
|
-
description:
|
|
312
|
-
context:
|
|
310
|
+
aliases: ["itl", "itul"],
|
|
311
|
+
description: "Kata vulgar yang mengacu pada bagian dari alat kelamin perempuan",
|
|
312
|
+
context: "Kata vulgar yang merujuk pada anatomi seksual",
|
|
313
313
|
},
|
|
314
314
|
{
|
|
315
|
-
word:
|
|
316
|
-
category:
|
|
317
|
-
region:
|
|
315
|
+
word: "kondom",
|
|
316
|
+
category: "sexual",
|
|
317
|
+
region: "general",
|
|
318
318
|
severity: 0.5,
|
|
319
|
-
aliases: [
|
|
320
|
-
description:
|
|
321
|
-
context:
|
|
319
|
+
aliases: ["kndm", "kondom", "cd"],
|
|
320
|
+
description: "Alat kontrasepsi",
|
|
321
|
+
context: "Dapat menjadi vulgar tergantung konteks penggunaan",
|
|
322
322
|
},
|
|
323
323
|
{
|
|
324
|
-
word:
|
|
325
|
-
category:
|
|
326
|
-
region:
|
|
324
|
+
word: "ngewe",
|
|
325
|
+
category: "sexual",
|
|
326
|
+
region: "general",
|
|
327
327
|
severity: 0.9,
|
|
328
|
-
aliases: [
|
|
329
|
-
description:
|
|
330
|
-
context:
|
|
328
|
+
aliases: ["ngew", "we"],
|
|
329
|
+
description: "Istilah kasar untuk aktivitas seksual",
|
|
330
|
+
context: "Kata vulgar yang merujuk pada aktivitas seksual",
|
|
331
331
|
},
|
|
332
332
|
{
|
|
333
|
-
word:
|
|
334
|
-
category:
|
|
335
|
-
region:
|
|
333
|
+
word: "puki",
|
|
334
|
+
category: "sexual",
|
|
335
|
+
region: "general",
|
|
336
336
|
severity: 0.9,
|
|
337
|
-
aliases: [
|
|
338
|
-
description:
|
|
339
|
-
context:
|
|
337
|
+
aliases: ["puk", "pukih"],
|
|
338
|
+
description: "Kata vulgar yang mengacu pada alat kelamin perempuan",
|
|
339
|
+
context: "Kata vulgar yang merujuk pada anatomi seksual",
|
|
340
340
|
},
|
|
341
341
|
{
|
|
342
|
-
word:
|
|
343
|
-
category:
|
|
344
|
-
region:
|
|
342
|
+
word: "xxx",
|
|
343
|
+
category: "sexual",
|
|
344
|
+
region: "general",
|
|
345
345
|
severity: 0.6,
|
|
346
|
-
aliases: [
|
|
347
|
-
description:
|
|
348
|
-
context:
|
|
346
|
+
aliases: ["xXx", "triplex"],
|
|
347
|
+
description: "Simbol yang sering digunakan untuk menandai konten pornografi",
|
|
348
|
+
context: "Digunakan untuk menandai konten seksual eksplisit",
|
|
349
349
|
},
|
|
350
350
|
];
|
|
351
351
|
sexual.map((item) => item.word);
|
|
352
352
|
|
|
353
353
|
const insult = [
|
|
354
|
-
...general.filter((word) => word.category ===
|
|
355
|
-
...jawa.filter((word) => word.category ===
|
|
356
|
-
...sunda.filter((word) => word.category ===
|
|
357
|
-
...betawi.filter((word) => word.category ===
|
|
354
|
+
...general.filter((word) => word.category === "insult"),
|
|
355
|
+
...jawa.filter((word) => word.category === "insult"),
|
|
356
|
+
...sunda.filter((word) => word.category === "insult"),
|
|
357
|
+
...betawi.filter((word) => word.category === "insult"),
|
|
358
358
|
{
|
|
359
|
-
word:
|
|
360
|
-
category:
|
|
361
|
-
region:
|
|
359
|
+
word: "idiot",
|
|
360
|
+
category: "insult",
|
|
361
|
+
region: "general",
|
|
362
362
|
severity: 0.6,
|
|
363
|
-
aliases: [
|
|
364
|
-
description:
|
|
365
|
-
context:
|
|
363
|
+
aliases: ["idi0t", "idot"],
|
|
364
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
365
|
+
context: "Hinaan untuk menyebut orang yang dianggap sangat tidak pintar",
|
|
366
366
|
},
|
|
367
367
|
{
|
|
368
|
-
word:
|
|
369
|
-
category:
|
|
370
|
-
region:
|
|
368
|
+
word: "dungu",
|
|
369
|
+
category: "insult",
|
|
370
|
+
region: "general",
|
|
371
371
|
severity: 0.5,
|
|
372
|
-
aliases: [
|
|
373
|
-
description:
|
|
374
|
-
context:
|
|
372
|
+
aliases: ["dngu", "dongu"],
|
|
373
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
374
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
375
375
|
},
|
|
376
376
|
{
|
|
377
|
-
word:
|
|
378
|
-
category:
|
|
379
|
-
region:
|
|
377
|
+
word: "sinting",
|
|
378
|
+
category: "insult",
|
|
379
|
+
region: "general",
|
|
380
380
|
severity: 0.6,
|
|
381
|
-
aliases: [
|
|
382
|
-
description:
|
|
383
|
-
context:
|
|
381
|
+
aliases: ["sintng", "senteng"],
|
|
382
|
+
description: "Kata yang mengacu pada kegilaan atau ketidakwarasan seseorang",
|
|
383
|
+
context: "Hinaan untuk menyebut orang yang dianggap gila atau tidak waras",
|
|
384
384
|
},
|
|
385
385
|
{
|
|
386
|
-
word:
|
|
387
|
-
category:
|
|
388
|
-
region:
|
|
386
|
+
word: "sarap",
|
|
387
|
+
category: "insult",
|
|
388
|
+
region: "general",
|
|
389
389
|
severity: 0.6,
|
|
390
|
-
aliases: [
|
|
391
|
-
description:
|
|
392
|
-
context:
|
|
390
|
+
aliases: ["saraf", "srap"],
|
|
391
|
+
description: "Kata yang mengacu pada kegilaan atau ketidakwarasan seseorang",
|
|
392
|
+
context: "Hinaan untuk menyebut orang yang dianggap gila atau tidak waras",
|
|
393
393
|
},
|
|
394
394
|
{
|
|
395
|
-
word:
|
|
396
|
-
category:
|
|
397
|
-
region:
|
|
395
|
+
word: "geblek",
|
|
396
|
+
category: "insult",
|
|
397
|
+
region: "general",
|
|
398
398
|
severity: 0.5,
|
|
399
|
-
aliases: [
|
|
400
|
-
description:
|
|
401
|
-
context:
|
|
399
|
+
aliases: ["gblk", "geblk"],
|
|
400
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
401
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
402
402
|
},
|
|
403
403
|
{
|
|
404
|
-
word:
|
|
405
|
-
category:
|
|
406
|
-
region:
|
|
404
|
+
word: "kampungan",
|
|
405
|
+
category: "insult",
|
|
406
|
+
region: "general",
|
|
407
407
|
severity: 0.5,
|
|
408
|
-
aliases: [
|
|
409
|
-
description:
|
|
410
|
-
context:
|
|
408
|
+
aliases: ["kmpngn", "kamphungan"],
|
|
409
|
+
description: "Kata yang mengacu pada ketidaksopanan atau kenaifan seseorang",
|
|
410
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak modern atau primitif",
|
|
411
411
|
},
|
|
412
412
|
{
|
|
413
|
-
word:
|
|
414
|
-
category:
|
|
415
|
-
region:
|
|
413
|
+
word: "udik",
|
|
414
|
+
category: "insult",
|
|
415
|
+
region: "general",
|
|
416
416
|
severity: 0.5,
|
|
417
|
-
aliases: [
|
|
418
|
-
description:
|
|
419
|
-
context:
|
|
417
|
+
aliases: ["udhik", "udek"],
|
|
418
|
+
description: "Kata yang mengacu pada ketidaksopanan atau kenaifan seseorang",
|
|
419
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak modern atau primitif",
|
|
420
420
|
},
|
|
421
421
|
{
|
|
422
|
-
word:
|
|
423
|
-
category:
|
|
424
|
-
region:
|
|
422
|
+
word: "dongo",
|
|
423
|
+
category: "insult",
|
|
424
|
+
region: "general",
|
|
425
425
|
severity: 0.5,
|
|
426
|
-
aliases: [
|
|
427
|
-
description:
|
|
428
|
-
context:
|
|
426
|
+
aliases: ["donggo", "dungu"],
|
|
427
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
428
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
429
429
|
},
|
|
430
430
|
{
|
|
431
|
-
word:
|
|
432
|
-
category:
|
|
433
|
-
region:
|
|
431
|
+
word: "tolol",
|
|
432
|
+
category: "insult",
|
|
433
|
+
region: "general",
|
|
434
434
|
severity: 0.6,
|
|
435
|
-
aliases: [
|
|
436
|
-
description:
|
|
437
|
-
context:
|
|
435
|
+
aliases: ["tol0l", "tollo", "tlol"],
|
|
436
|
+
description: "Kata yang mengacu pada kebodohan seseorang",
|
|
437
|
+
context: "Hinaan untuk menyebut orang yang dianggap tidak pintar",
|
|
438
438
|
},
|
|
439
439
|
{
|
|
440
|
-
word:
|
|
441
|
-
category:
|
|
442
|
-
region:
|
|
440
|
+
word: "brengsek",
|
|
441
|
+
category: "insult",
|
|
442
|
+
region: "general",
|
|
443
443
|
severity: 0.7,
|
|
444
|
-
aliases: [
|
|
445
|
-
description:
|
|
446
|
-
context:
|
|
444
|
+
aliases: ["brengskek", "brengsik", "brengzek"],
|
|
445
|
+
description: "Kata yang mengacu pada kelakuan buruk atau tidak bermoral",
|
|
446
|
+
context: "Hinaan untuk menyebut orang yang dianggap memiliki kelakuan buruk",
|
|
447
447
|
},
|
|
448
448
|
];
|
|
449
449
|
insult.map((item) => item.word);
|
|
450
450
|
|
|
451
451
|
const batak = [
|
|
452
452
|
{
|
|
453
|
-
word:
|
|
454
|
-
category:
|
|
455
|
-
region:
|
|
453
|
+
word: "sundel",
|
|
454
|
+
category: "sexual",
|
|
455
|
+
region: "batak",
|
|
456
456
|
severity: 0.8,
|
|
457
|
-
aliases: [
|
|
458
|
-
description:
|
|
459
|
-
context:
|
|
457
|
+
aliases: ["sundal"],
|
|
458
|
+
description: "Kata kasar untuk menyebut pekerja seks komersial",
|
|
459
|
+
context: "Hinaan kasar untuk wanita",
|
|
460
460
|
},
|
|
461
461
|
];
|
|
462
462
|
batak.map((item) => item.word);
|
|
@@ -489,10 +489,651 @@ wordObjects
|
|
|
489
489
|
.filter((word) => word.severity >= 0.8)
|
|
490
490
|
.map((item) => item.word);
|
|
491
491
|
|
|
492
|
+
/**
|
|
493
|
+
* Menyensor kata dengan karakter pengganti
|
|
494
|
+
*
|
|
495
|
+
* @param word Kata yang akan disensor
|
|
496
|
+
* @param replaceChar Karakter pengganti (default: '*')
|
|
497
|
+
* @param keepFirstAndLast Apakah harus menyimpan huruf pertama dan terakhir
|
|
498
|
+
* @returns Kata yang telah disensor
|
|
499
|
+
*/
|
|
500
|
+
function censorWord(word, replaceChar = "*", keepFirstAndLast = false) {
|
|
501
|
+
if (word.length <= 2) {
|
|
502
|
+
return replaceChar.repeat(word.length);
|
|
503
|
+
}
|
|
504
|
+
if (keepFirstAndLast) {
|
|
505
|
+
return `${word[0]}${replaceChar.repeat(word.length - 2)}${word[word.length - 1]}`;
|
|
506
|
+
}
|
|
507
|
+
return replaceChar.repeat(word.length);
|
|
508
|
+
}
|
|
509
|
+
/**
|
|
510
|
+
* Menormalisasi string untuk keperluan pembandingan
|
|
511
|
+
*
|
|
512
|
+
* @param text Teks yang akan dinormalisasi
|
|
513
|
+
* @returns Teks yang telah dinormalisasi
|
|
514
|
+
*/
|
|
515
|
+
function normalizeText(text) {
|
|
516
|
+
return text
|
|
517
|
+
.toLowerCase()
|
|
518
|
+
.normalize("NFD") // Normalisasi Unicode
|
|
519
|
+
.replace(/[\u0300-\u036f]/g, "") // Hapus diacritic marks
|
|
520
|
+
.replace(/[^\w\s]/g, "") // Hapus karakter non-alphanumeric
|
|
521
|
+
.trim(); // Hapus whitespace di awal dan akhir
|
|
522
|
+
}
|
|
523
|
+
/**
|
|
524
|
+
* Mendeteksi apakah string berisi sebagian atau keseluruhan kata dalam wordList
|
|
525
|
+
*
|
|
526
|
+
* @param text Teks yang akan diperiksa
|
|
527
|
+
* @param wordList Daftar kata yang dicari
|
|
528
|
+
* @param checkSubstring Apakah harus memeriksa substring
|
|
529
|
+
* @returns Boolean apakah teks mengandung kata-kata dalam wordList
|
|
530
|
+
*/
|
|
531
|
+
function containsAnyWord(text, wordList, checkSubstring = false) {
|
|
532
|
+
const normalizedText = normalizeText(text);
|
|
533
|
+
return wordList.some((word) => {
|
|
534
|
+
const normalizedWord = normalizeText(word);
|
|
535
|
+
return checkSubstring
|
|
536
|
+
? normalizedText.includes(normalizedWord)
|
|
537
|
+
: new RegExp(`\\b${escapeRegExp(normalizedWord)}\\b`, "i").test(normalizedText);
|
|
538
|
+
});
|
|
539
|
+
}
|
|
540
|
+
/**
|
|
541
|
+
* Memerikasa apakah stirng merupakan kode untuk kata kotor
|
|
542
|
+
* (Menangkap kasus seperti disensor dengan titik atau garis: a**ing, b*bi, dll)
|
|
543
|
+
*
|
|
544
|
+
* @param text Teks yang akan diperika
|
|
545
|
+
* @param wordList daftar kata kotor
|
|
546
|
+
* @return Boolean apakah teks mengandung kata kotor
|
|
547
|
+
*/
|
|
548
|
+
function containsEuphemism(text, wordList) {
|
|
549
|
+
return wordList.some((word) => {
|
|
550
|
+
if (word.length <= 2)
|
|
551
|
+
return false;
|
|
552
|
+
const firstChar = word[0];
|
|
553
|
+
const lastChar = word[word.length - 1];
|
|
554
|
+
const pattern = new RegExp(`\\b${escapeRegExp(firstChar)}[*@#\\-_.!?\\s]{${word.length - 2}}${escapeRegExp(lastChar)}\\b`, "i");
|
|
555
|
+
return pattern.test(text);
|
|
556
|
+
});
|
|
557
|
+
}
|
|
558
|
+
/**
|
|
559
|
+
* Mendeteksi upaya menghindari filter dengan pemisahan kata
|
|
560
|
+
* (Tangkap kasus seperti: a n j i n g, b-a-b-i, dll)
|
|
561
|
+
*
|
|
562
|
+
* @param text Teks yang akan diperiksa
|
|
563
|
+
* @param wordList Daftar kata kotor
|
|
564
|
+
* @returns Boolean apakah teks mengandung upaya menghindari filter
|
|
565
|
+
*/
|
|
566
|
+
function detectSplitWords(text, wordList) {
|
|
567
|
+
const compressedText = text.replace(/[\s\-_.!?*]/g, "").toLowerCase();
|
|
568
|
+
return wordList.some((word) => compressedText.includes(normalizeText(word)));
|
|
569
|
+
}
|
|
570
|
+
function escapeRegExp(string) {
|
|
571
|
+
return string.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
572
|
+
}
|
|
573
|
+
/**
|
|
574
|
+
* Mengganti sebagian kata dengan masking
|
|
575
|
+
* (berguna untuk email, nomor telepon, dll)
|
|
576
|
+
*
|
|
577
|
+
* @param text Teks untuk dimasking
|
|
578
|
+
* @param visibleStart Jumlah karakter yang terlihat di awal
|
|
579
|
+
* @param visibleEnd Jumlah karakter yang terlihat di akhir
|
|
580
|
+
* @param maskChar Karakter masking
|
|
581
|
+
* @returns Teks yang telah dimasking
|
|
582
|
+
*/
|
|
583
|
+
function maskText(text, visibleStart = 1, visibleEnd = 1, maskChar = "*") {
|
|
584
|
+
if (!text)
|
|
585
|
+
return "";
|
|
586
|
+
if (text.length <= visibleStart + visibleEnd)
|
|
587
|
+
return text;
|
|
588
|
+
const start = text.substring(0, visibleStart);
|
|
589
|
+
const middle = maskChar.repeat(text.length - visibleStart - visibleEnd);
|
|
590
|
+
const end = text.substring(text.length - visibleEnd);
|
|
591
|
+
return start + middle + end;
|
|
592
|
+
}
|
|
593
|
+
/**
|
|
594
|
+
* menguabh sting menjadi bentuk leet speak
|
|
595
|
+
* (untuk testing filter bypass)
|
|
596
|
+
*
|
|
597
|
+
* @param text Teks yang akan diubah
|
|
598
|
+
* @return Teks yang telah diubah ke leet speak
|
|
599
|
+
*/
|
|
600
|
+
function toLeetSpeak(text) {
|
|
601
|
+
const leetMap = {
|
|
602
|
+
a: ["4", "@"],
|
|
603
|
+
b: ["8", "6"],
|
|
604
|
+
c: ["<", "(", "{"],
|
|
605
|
+
e: ["3"],
|
|
606
|
+
g: ["9"],
|
|
607
|
+
i: ["1", "!"],
|
|
608
|
+
l: ["1", "|"],
|
|
609
|
+
o: ["0"],
|
|
610
|
+
s: ["5", "$"],
|
|
611
|
+
t: ["7", "+"],
|
|
612
|
+
z: ["2"],
|
|
613
|
+
};
|
|
614
|
+
return text
|
|
615
|
+
.split("")
|
|
616
|
+
.map((char) => {
|
|
617
|
+
const lowerChar = char.toLowerCase();
|
|
618
|
+
return leetMap[lowerChar] || char;
|
|
619
|
+
})
|
|
620
|
+
.join("");
|
|
621
|
+
}
|
|
622
|
+
/**
|
|
623
|
+
* memisahkan teks menajdi kelimat
|
|
624
|
+
*
|
|
625
|
+
* @param text Teks yang akan dipisahkan
|
|
626
|
+
* @return Array kalimat yang telah dipisahkan
|
|
627
|
+
*/
|
|
628
|
+
function splitIntoSentences(text) {
|
|
629
|
+
// split berdasarkan titik, seru, taya yagn diikuti spasi atau akhir string
|
|
630
|
+
return text
|
|
631
|
+
.split(/(?<=[.!?])\s+|(?<=[.!?])$/)
|
|
632
|
+
.filter((sentence) => sentence.trim().length > 0);
|
|
633
|
+
}
|
|
634
|
+
/**
|
|
635
|
+
* mengambil kata-kata di sekitar indeks tertentu
|
|
636
|
+
*
|
|
637
|
+
* @param text Teks yang akan diambil
|
|
638
|
+
* @param index Indeks dalam teks
|
|
639
|
+
* @param windowSize jumlah kata di sekitar indeks
|
|
640
|
+
* @return Kata-kata di sekitar indeks
|
|
641
|
+
*/
|
|
642
|
+
function getContextAroundIndex(text, index, windowSize = 5) {
|
|
643
|
+
if (!text || index < 0 || index >= text.length)
|
|
644
|
+
return "";
|
|
645
|
+
const words = text.split(/\s+/);
|
|
646
|
+
let currentPosition = 0;
|
|
647
|
+
let targetWordIndex = -1;
|
|
648
|
+
for (let i = 0; i < words.length; i++) {
|
|
649
|
+
const wordLength = words[i].length;
|
|
650
|
+
if (index >= currentPosition && index < currentPosition + wordLength) {
|
|
651
|
+
targetWordIndex = i;
|
|
652
|
+
break;
|
|
653
|
+
}
|
|
654
|
+
// Tambahkan panjang kata dan spasi
|
|
655
|
+
currentPosition += wordLength + 1;
|
|
656
|
+
}
|
|
657
|
+
if (targetWordIndex === -1)
|
|
658
|
+
return "";
|
|
659
|
+
// Ambil kata-kata di sekitar
|
|
660
|
+
const startIndex = Math.max(0, targetWordIndex - windowSize);
|
|
661
|
+
const endIndex = Math.min(words.length, targetWordIndex + windowSize + 1);
|
|
662
|
+
return words.slice(startIndex, endIndex).join(" ");
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
/**
|
|
666
|
+
* Membuat pola regex untuk mencocokkan kata
|
|
667
|
+
*
|
|
668
|
+
* @param word Kata yang akan dibuat pola regex-nya
|
|
669
|
+
* @param options Opsi untuk pembuatan regex
|
|
670
|
+
* @returns Objek RegExp
|
|
671
|
+
*/
|
|
672
|
+
function createWordRegex(word, options = {}) {
|
|
673
|
+
const { wholeWord = true, caseSensitive = false, leetSpeak = true, detectSplit = false, indonesianVariation = false, } = options;
|
|
674
|
+
// Escape karakter khusus regex
|
|
675
|
+
let pattern = escapeRegExp(word);
|
|
676
|
+
// Tambahkan variasi leet speak jika diminta
|
|
677
|
+
if (leetSpeak) {
|
|
678
|
+
pattern = addLeetSpeakVariations(pattern);
|
|
679
|
+
}
|
|
680
|
+
// Tambahkan variasi ejaan Bahasa Indonesia jika diminta
|
|
681
|
+
if (indonesianVariation) {
|
|
682
|
+
pattern = addIndonesianVariations(pattern);
|
|
683
|
+
}
|
|
684
|
+
// Tambahkan kemungkinan split jika diminta
|
|
685
|
+
if (detectSplit) {
|
|
686
|
+
pattern = addSplitVariations(pattern);
|
|
687
|
+
}
|
|
688
|
+
// Tambahkan boundary untuk whole word jika diminta
|
|
689
|
+
if (wholeWord) {
|
|
690
|
+
pattern = `\\b${pattern}\\b`;
|
|
691
|
+
}
|
|
692
|
+
// Buat regex dengan flag case-insensitive jika diminta
|
|
693
|
+
return new RegExp(pattern, caseSensitive ? "g" : "gi");
|
|
694
|
+
}
|
|
695
|
+
/**
|
|
696
|
+
* Menambahkan variasi leet speak ke pola regex
|
|
697
|
+
*
|
|
698
|
+
* Contoh:
|
|
699
|
+
* - 'a' bisa jadi '4', '@'
|
|
700
|
+
* - 'i' bisa jadi '1', '!'
|
|
701
|
+
*
|
|
702
|
+
* @param pattern Pola regex asli
|
|
703
|
+
* @returns Pola regex dengan variasi leet speak
|
|
704
|
+
*/
|
|
705
|
+
function addLeetSpeakVariations(pattern) {
|
|
706
|
+
const leetMap = {
|
|
707
|
+
a: ["a", "4", "@"],
|
|
708
|
+
b: ["b", "8", "6"],
|
|
709
|
+
c: ["c", "(", "{", "<"],
|
|
710
|
+
e: ["e", "3"],
|
|
711
|
+
g: ["g", "6", "9"],
|
|
712
|
+
i: ["i", "1", "!", "|"],
|
|
713
|
+
l: ["l", "1", "|"],
|
|
714
|
+
o: ["o", "0"],
|
|
715
|
+
s: ["s", "5", "$"],
|
|
716
|
+
t: ["t", "7", "+"],
|
|
717
|
+
z: ["z", "2"],
|
|
718
|
+
};
|
|
719
|
+
// Ganti tiap karakter dengan variasinya dalam grup character class
|
|
720
|
+
return pattern
|
|
721
|
+
.split("")
|
|
722
|
+
.map((char) => {
|
|
723
|
+
const lowerChar = char.toLowerCase();
|
|
724
|
+
const variations = leetMap[lowerChar];
|
|
725
|
+
if (variations && variations.length > 1) {
|
|
726
|
+
return `[${variations.join("")}]`;
|
|
727
|
+
}
|
|
728
|
+
return char;
|
|
729
|
+
})
|
|
730
|
+
.join("");
|
|
731
|
+
}
|
|
732
|
+
/**
|
|
733
|
+
* Menambahkan kemungkinan split/pemisahan antar karakter
|
|
734
|
+
*
|
|
735
|
+
* @param pattern Pola regex asli
|
|
736
|
+
* @returns Pola regex dengan kemungkinan split
|
|
737
|
+
*/
|
|
738
|
+
function addSplitVariations(pattern) {
|
|
739
|
+
// Tambahkan kemungkinan spasi atau karakter penghubung di antara setiap huruf
|
|
740
|
+
return pattern.split("").join("[\\s\\-._*+]?");
|
|
741
|
+
}
|
|
742
|
+
/**
|
|
743
|
+
* Membuat regex untuk mencari kata dengan variasi spasi dan karakter penghubung
|
|
744
|
+
* Berguna untuk mendeteksi upaya menghindari filter dengan menambahkan spasi atau karakter lain
|
|
745
|
+
*
|
|
746
|
+
* @param word Kata yang akan dibuat pola regexnya
|
|
747
|
+
* @returns Objek RegExp
|
|
748
|
+
*/
|
|
749
|
+
function createEvasionRegex(word) {
|
|
750
|
+
// Tambahkan kemungkinan spasi atau karakter penghubung di antara setiap huruf
|
|
751
|
+
const pattern = addSplitVariations(escapeRegExp(word));
|
|
752
|
+
return new RegExp(pattern, "gi");
|
|
753
|
+
}
|
|
754
|
+
/**
|
|
755
|
+
* Menambahkan variasi ejaan Bahasa Indonesia
|
|
756
|
+
*
|
|
757
|
+
* @param pattern Pola regex asli
|
|
758
|
+
* @returns Pola regex dengan variasi ejaan Bahasa Indonesia
|
|
759
|
+
*/
|
|
760
|
+
function addIndonesianVariations(pattern) {
|
|
761
|
+
// Variasi ejaan dalam Bahasa Indonesia
|
|
762
|
+
const variationMap = {
|
|
763
|
+
c: ["c", "k"], // contoh: becok/bekok
|
|
764
|
+
k: ["k", "c", "q"], // contoh: kacau/qacau
|
|
765
|
+
j: ["j", "dj"], // contoh: jualan/djualan (ejaan lama)
|
|
766
|
+
y: ["y", "j"], // contoh: ya/ja
|
|
767
|
+
u: ["u", "oe"], // contoh: untuk/oentoek (ejaan lama)
|
|
768
|
+
f: ["f", "p", "v"], // contoh: kafir/kapir
|
|
769
|
+
z: ["z", "j", "s"], // contoh: zaman/jaman
|
|
770
|
+
x: ["x", "ks"], // contoh: taxi/taksi
|
|
771
|
+
};
|
|
772
|
+
// Ganti tiap karakter dengan variasinya
|
|
773
|
+
return pattern
|
|
774
|
+
.split("")
|
|
775
|
+
.map((char) => {
|
|
776
|
+
const lowerChar = char.toLowerCase();
|
|
777
|
+
const variations = variationMap[lowerChar];
|
|
778
|
+
if (variations && variations.length > 1) {
|
|
779
|
+
return `[${variations.join("")}]`;
|
|
780
|
+
}
|
|
781
|
+
return char;
|
|
782
|
+
})
|
|
783
|
+
.join("");
|
|
784
|
+
}
|
|
785
|
+
/**
|
|
786
|
+
* Membuat regex untuk mencocokkan kata dengan mempertimbangkan variasi ejaan Bahasa Indonesia
|
|
787
|
+
*
|
|
788
|
+
* @param word Kata yang akan dibuat pola regexnya
|
|
789
|
+
* @returns Objek RegExp
|
|
790
|
+
*/
|
|
791
|
+
function createIndonesianVariationRegex(word) {
|
|
792
|
+
const pattern = addIndonesianVariations(escapeRegExp(word));
|
|
793
|
+
return new RegExp(`\\b${pattern}\\b`, "gi");
|
|
794
|
+
}
|
|
795
|
+
/**
|
|
796
|
+
* Membuat regex untuk mencocokkan kata dengan konteks
|
|
797
|
+
*
|
|
798
|
+
* @param word Kata yang akan dibuat pola regexnya
|
|
799
|
+
* @param contextSize Jumlah kata konteks sebelum dan sesudah
|
|
800
|
+
* @returns Objek RegExp
|
|
801
|
+
*/
|
|
802
|
+
function createContextRegex(word, contextSize = 3) {
|
|
803
|
+
const wordPattern = escapeRegExp(word);
|
|
804
|
+
// Membuat pola yang menangkap beberapa kata sebelum dan setelah kata target
|
|
805
|
+
const pattern = `((?:\\S+\\s+){0,${contextSize}})(\\b${wordPattern}\\b)((?:\\s+\\S+){0,${contextSize}})`;
|
|
806
|
+
return new RegExp(pattern, "gi");
|
|
807
|
+
}
|
|
808
|
+
/**
|
|
809
|
+
* Membuat regex untuk mencocokkan variasi penulisan kata
|
|
810
|
+
*
|
|
811
|
+
* @param word Kata dasar
|
|
812
|
+
* @returns Objek RegExp untuk mencocokkan berbagai bentuk kata
|
|
813
|
+
*/
|
|
814
|
+
function createWordFormRegex(word) {
|
|
815
|
+
// Implementasi sederhana untuk mencocokkan berbagai imbuhan
|
|
816
|
+
// Ini bisa dikembangkan lebih lanjut untuk mencocokkan bentukan kata yang lebih kompleks
|
|
817
|
+
const prefixes = ["", "me", "pe", "ber", "di", "ter", "se"];
|
|
818
|
+
const suffixes = ["", "kan", "an", "i", "nya"];
|
|
819
|
+
const patterns = [];
|
|
820
|
+
// Kombinasikan prefix dan suffix
|
|
821
|
+
for (const prefix of prefixes) {
|
|
822
|
+
for (const suffix of suffixes) {
|
|
823
|
+
patterns.push(`\\b${prefix}${escapeRegExp(word)}${suffix}\\b`);
|
|
824
|
+
}
|
|
825
|
+
}
|
|
826
|
+
return new RegExp(patterns.join("|"), "gi");
|
|
827
|
+
}
|
|
828
|
+
|
|
829
|
+
/**
|
|
830
|
+
* Menghitung jarak Levenshtein antara dua string
|
|
831
|
+
* (Jumlah operasi insert, delete, atau replace untuk mengubah string1 menjadi string2)
|
|
832
|
+
*
|
|
833
|
+
* @param str1 String pertama
|
|
834
|
+
* @param str2 String kedua
|
|
835
|
+
* @returns Jarak Levenshtein
|
|
836
|
+
*/
|
|
837
|
+
function levenshteinDistance(str1, str2) {
|
|
838
|
+
const s1 = str1.toLowerCase();
|
|
839
|
+
const s2 = str2.toLowerCase();
|
|
840
|
+
const len1 = s1.length;
|
|
841
|
+
const len2 = s2.length;
|
|
842
|
+
// Inisialisasi matrix
|
|
843
|
+
const matrix = [];
|
|
844
|
+
// Inisialisasi baris pertama
|
|
845
|
+
for (let i = 0; i <= len2; i++) {
|
|
846
|
+
matrix[0] = matrix[0] || [];
|
|
847
|
+
matrix[0][i] = i;
|
|
848
|
+
}
|
|
849
|
+
// Inisialisasi kolom pertama
|
|
850
|
+
for (let i = 0; i <= len1; i++) {
|
|
851
|
+
matrix[i] = matrix[i] || [];
|
|
852
|
+
matrix[i][0] = i;
|
|
853
|
+
}
|
|
854
|
+
// Isi matrix
|
|
855
|
+
for (let i = 1; i <= len1; i++) {
|
|
856
|
+
for (let j = 1; j <= len2; j++) {
|
|
857
|
+
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
|
858
|
+
matrix[i][j] = Math.min(matrix[i - 1][j] + 1, // deletion
|
|
859
|
+
matrix[i][j - 1] + 1, // insertion
|
|
860
|
+
matrix[i - 1][j - 1] + cost);
|
|
861
|
+
}
|
|
862
|
+
}
|
|
863
|
+
return matrix[len1][len2];
|
|
864
|
+
}
|
|
865
|
+
/**
|
|
866
|
+
* Menghitung tingkat kesamaan antara dua string
|
|
867
|
+
*
|
|
868
|
+
* @param str1 String pertama
|
|
869
|
+
* @param str2 String kedua
|
|
870
|
+
* @returns Nilai kesamaan (0-1, di mana 1 berarti identik)
|
|
871
|
+
*/
|
|
872
|
+
function stringSimilarity(str1, str2) {
|
|
873
|
+
if (!str1.length && !str2.length)
|
|
874
|
+
return 1;
|
|
875
|
+
if (!str1.length || !str2.length)
|
|
876
|
+
return 0;
|
|
877
|
+
const distance = levenshteinDistance(str1, str2);
|
|
878
|
+
const maxLength = Math.max(str1.length, str2.length);
|
|
879
|
+
return 1 - distance / maxLength;
|
|
880
|
+
}
|
|
881
|
+
/**
|
|
882
|
+
* Mencari string yang paling mirip dari array
|
|
883
|
+
*
|
|
884
|
+
* @param target String target
|
|
885
|
+
* @param candidates Array string kandidat
|
|
886
|
+
* @param threshold Minimum kesamaan yang diterima (0-1)
|
|
887
|
+
* @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
|
|
888
|
+
*/
|
|
889
|
+
function findMostSimilar(target, candidates, threshold = 0.7) {
|
|
890
|
+
if (!candidates.length)
|
|
891
|
+
return null;
|
|
892
|
+
let maxSimilarity = 0;
|
|
893
|
+
let mostSimilar = null;
|
|
894
|
+
for (const candidate of candidates) {
|
|
895
|
+
const similarity = stringSimilarity(target, candidate);
|
|
896
|
+
if (similarity > maxSimilarity && similarity >= threshold) {
|
|
897
|
+
maxSimilarity = similarity;
|
|
898
|
+
mostSimilar = candidate;
|
|
899
|
+
}
|
|
900
|
+
}
|
|
901
|
+
return mostSimilar;
|
|
902
|
+
}
|
|
903
|
+
/**
|
|
904
|
+
* Cek apakah string mungkin merupakan variasi dari kata kotor
|
|
905
|
+
* menggunakan kesamaan string
|
|
906
|
+
*
|
|
907
|
+
* @param input String yang akan diperiksa
|
|
908
|
+
* @param profanityWords Daftar kata kotor
|
|
909
|
+
* @param threshold Batas minimum kesamaan (default: 0.75)
|
|
910
|
+
* @returns Array [Boolean (apakah variasi), String original (jika ditemukan)]
|
|
911
|
+
*/
|
|
912
|
+
function isPossibleProfanityVariation(input, profanityWords, threshold = 0.75) {
|
|
913
|
+
if (!input || !profanityWords.length)
|
|
914
|
+
return [false, null];
|
|
915
|
+
for (const word of profanityWords) {
|
|
916
|
+
const similarity = stringSimilarity(input, word);
|
|
917
|
+
if (similarity >= threshold) {
|
|
918
|
+
return [true, word];
|
|
919
|
+
}
|
|
920
|
+
}
|
|
921
|
+
return [false, null];
|
|
922
|
+
}
|
|
923
|
+
/**
|
|
924
|
+
* Mengelompokkan kata berdasarkan kesamaan
|
|
925
|
+
*
|
|
926
|
+
* @param words Daftar kata
|
|
927
|
+
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
928
|
+
* @returns Array kluster kata yang mirip
|
|
929
|
+
*/
|
|
930
|
+
function clusterSimilarWords(words, threshold = 0.8) {
|
|
931
|
+
const clusters = [];
|
|
932
|
+
const processed = new Set();
|
|
933
|
+
for (const word of words) {
|
|
934
|
+
if (processed.has(word))
|
|
935
|
+
continue;
|
|
936
|
+
const cluster = [word];
|
|
937
|
+
processed.add(word);
|
|
938
|
+
for (const otherWord of words) {
|
|
939
|
+
if (word === otherWord || processed.has(otherWord))
|
|
940
|
+
continue;
|
|
941
|
+
const similarity = stringSimilarity(word, otherWord);
|
|
942
|
+
if (similarity >= threshold) {
|
|
943
|
+
cluster.push(otherWord);
|
|
944
|
+
processed.add(otherWord);
|
|
945
|
+
}
|
|
946
|
+
}
|
|
947
|
+
clusters.push(cluster);
|
|
948
|
+
}
|
|
949
|
+
return clusters;
|
|
950
|
+
}
|
|
951
|
+
/**
|
|
952
|
+
* Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
|
|
953
|
+
*
|
|
954
|
+
* @param text Teks yang akan diperiksa
|
|
955
|
+
* @param profanityWords Daftar kata kotor
|
|
956
|
+
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
957
|
+
* @returns Array kata yang mungkin merupakan kata kotor
|
|
958
|
+
*/
|
|
959
|
+
function findPossibleProfanityBySimiliarity(text, profanityWords, threshold = 0.8) {
|
|
960
|
+
const result = [];
|
|
961
|
+
// Pisahkan teks menjadi kata-kata
|
|
962
|
+
const words = text.toLowerCase().split(/\s+/);
|
|
963
|
+
for (const word of words) {
|
|
964
|
+
// Lewati kata-kata yang terlalu pendek
|
|
965
|
+
if (word.length < 3)
|
|
966
|
+
continue;
|
|
967
|
+
for (const profanity of profanityWords) {
|
|
968
|
+
const similarity = stringSimilarity(word, profanity);
|
|
969
|
+
if (similarity >= threshold) {
|
|
970
|
+
result.push({
|
|
971
|
+
word,
|
|
972
|
+
original: profanity,
|
|
973
|
+
similarity,
|
|
974
|
+
});
|
|
975
|
+
break;
|
|
976
|
+
}
|
|
977
|
+
}
|
|
978
|
+
}
|
|
979
|
+
return result;
|
|
980
|
+
}
|
|
981
|
+
|
|
982
|
+
const DEFAULT_OPTIONS = {
|
|
983
|
+
replaceWith: "*",
|
|
984
|
+
fullWordCensor: true,
|
|
985
|
+
detectLeetSpeak: true,
|
|
986
|
+
checkSubstring: false,
|
|
987
|
+
whitelist: [],
|
|
988
|
+
severityThreshold: 0,
|
|
989
|
+
};
|
|
990
|
+
const FILTER_PRESETS = {
|
|
991
|
+
strict: {
|
|
992
|
+
...DEFAULT_OPTIONS,
|
|
993
|
+
checkSubstring: true,
|
|
994
|
+
detectLeetSpeak: true,
|
|
995
|
+
severityThreshold: 0,
|
|
996
|
+
},
|
|
997
|
+
moderate: {
|
|
998
|
+
...DEFAULT_OPTIONS,
|
|
999
|
+
severityThreshold: 0.5,
|
|
1000
|
+
},
|
|
1001
|
+
light: {
|
|
1002
|
+
...DEFAULT_OPTIONS,
|
|
1003
|
+
severityThreshold: 0.7,
|
|
1004
|
+
categories: ["sexual", "slur", "blasphemy"],
|
|
1005
|
+
},
|
|
1006
|
+
childSafe: {
|
|
1007
|
+
...DEFAULT_OPTIONS,
|
|
1008
|
+
checkSubstring: true,
|
|
1009
|
+
detectLeetSpeak: true,
|
|
1010
|
+
fullWordCensor: true,
|
|
1011
|
+
severityThreshold: 0,
|
|
1012
|
+
},
|
|
1013
|
+
};
|
|
1014
|
+
const CATEGORY_PRESETS = {
|
|
1015
|
+
sexual: {
|
|
1016
|
+
...DEFAULT_OPTIONS,
|
|
1017
|
+
categories: ["sexual"],
|
|
1018
|
+
},
|
|
1019
|
+
insults: {
|
|
1020
|
+
...DEFAULT_OPTIONS,
|
|
1021
|
+
categories: ["insult"],
|
|
1022
|
+
},
|
|
1023
|
+
profanity: {
|
|
1024
|
+
...DEFAULT_OPTIONS,
|
|
1025
|
+
categories: ["profanity"],
|
|
1026
|
+
},
|
|
1027
|
+
};
|
|
1028
|
+
const REGION_PRESETS = {
|
|
1029
|
+
general: {
|
|
1030
|
+
...DEFAULT_OPTIONS,
|
|
1031
|
+
regions: ["general"],
|
|
1032
|
+
},
|
|
1033
|
+
jawa: {
|
|
1034
|
+
...DEFAULT_OPTIONS,
|
|
1035
|
+
regions: ["jawa"],
|
|
1036
|
+
},
|
|
1037
|
+
sunda: {
|
|
1038
|
+
...DEFAULT_OPTIONS,
|
|
1039
|
+
regions: ["sunda"],
|
|
1040
|
+
},
|
|
1041
|
+
betawi: {
|
|
1042
|
+
...DEFAULT_OPTIONS,
|
|
1043
|
+
regions: ["betawi"],
|
|
1044
|
+
},
|
|
1045
|
+
batak: {
|
|
1046
|
+
...DEFAULT_OPTIONS,
|
|
1047
|
+
regions: ["batak"],
|
|
1048
|
+
},
|
|
1049
|
+
};
|
|
1050
|
+
const REPLACEMENT_CHARS = {
|
|
1051
|
+
asterisk: "*",
|
|
1052
|
+
hash: "#",
|
|
1053
|
+
dollar: "$",
|
|
1054
|
+
at: "@",
|
|
1055
|
+
percent: "%",
|
|
1056
|
+
underscore: "_",
|
|
1057
|
+
dash: "-",
|
|
1058
|
+
dot: ".",
|
|
1059
|
+
grawlix: "#@$%&!",
|
|
1060
|
+
};
|
|
1061
|
+
/**
|
|
1062
|
+
* Membuat opsi custom dengan menggabungkan dengan default
|
|
1063
|
+
*
|
|
1064
|
+
* @param options Opsi yang akan digabungkan dengan default
|
|
1065
|
+
* @returns Opsi yang sudah digabungkan
|
|
1066
|
+
*/
|
|
1067
|
+
function createOptions(options = {}) {
|
|
1068
|
+
return {
|
|
1069
|
+
...DEFAULT_OPTIONS,
|
|
1070
|
+
...options,
|
|
1071
|
+
};
|
|
1072
|
+
}
|
|
1073
|
+
/**
|
|
1074
|
+
* Mendapatkan opsi dari preset yang ada
|
|
1075
|
+
*
|
|
1076
|
+
* @param presetName Nama preset filter, kategori, atau region
|
|
1077
|
+
* @param additionalOptions Opsi tambahan untuk mengganti preset
|
|
1078
|
+
* @returns Opsi yang sudah digabungkan
|
|
1079
|
+
*/
|
|
1080
|
+
function getPresetOptions(presetName, additionalOptions = {}) {
|
|
1081
|
+
let presetOptions;
|
|
1082
|
+
if (presetName in FILTER_PRESETS) {
|
|
1083
|
+
presetOptions = FILTER_PRESETS[presetName];
|
|
1084
|
+
}
|
|
1085
|
+
else if (presetName in CATEGORY_PRESETS) {
|
|
1086
|
+
presetOptions =
|
|
1087
|
+
CATEGORY_PRESETS[presetName];
|
|
1088
|
+
}
|
|
1089
|
+
else if (presetName in REGION_PRESETS) {
|
|
1090
|
+
presetOptions = REGION_PRESETS[presetName];
|
|
1091
|
+
}
|
|
1092
|
+
else {
|
|
1093
|
+
presetOptions = FILTER_PRESETS.strict;
|
|
1094
|
+
}
|
|
1095
|
+
return {
|
|
1096
|
+
...presetOptions,
|
|
1097
|
+
...additionalOptions,
|
|
1098
|
+
};
|
|
1099
|
+
}
|
|
1100
|
+
/**
|
|
1101
|
+
* Mendapatkan karakter pengganti
|
|
1102
|
+
*
|
|
1103
|
+
* @param type Tipe karakter pengganti
|
|
1104
|
+
* @returns Karakter pengganti
|
|
1105
|
+
*/
|
|
1106
|
+
function getReplacementChar(type = "asterisk") {
|
|
1107
|
+
return REPLACEMENT_CHARS[type] || REPLACEMENT_CHARS.asterisk;
|
|
1108
|
+
}
|
|
1109
|
+
/**
|
|
1110
|
+
* Membuat karakter pengganti random dari grawlix
|
|
1111
|
+
*
|
|
1112
|
+
* @returns Karakter pengganti random
|
|
1113
|
+
*/
|
|
1114
|
+
function getRandomGrawlix() {
|
|
1115
|
+
const grawlix = REPLACEMENT_CHARS.grawlix;
|
|
1116
|
+
return grawlix[Math.floor(Math.random() * grawlix.length)];
|
|
1117
|
+
}
|
|
1118
|
+
/**
|
|
1119
|
+
* Membuat string pengganti untuk kata menggunakan grawlix random
|
|
1120
|
+
*
|
|
1121
|
+
* @param length Panjang string
|
|
1122
|
+
* @returns String pengganti
|
|
1123
|
+
*/
|
|
1124
|
+
function makeRandomGrawlixString(length) {
|
|
1125
|
+
let result = "";
|
|
1126
|
+
const grawlix = REPLACEMENT_CHARS.grawlix;
|
|
1127
|
+
for (let i = 0; i < length; i++) {
|
|
1128
|
+
result += grawlix[Math.floor(Math.random() * grawlix.length)];
|
|
1129
|
+
}
|
|
1130
|
+
return result;
|
|
1131
|
+
}
|
|
1132
|
+
|
|
492
1133
|
function findProfanity(text, options = {}) {
|
|
493
|
-
const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, } = options;
|
|
1134
|
+
const { wordList = [], detectLeetSpeak = true, checkSubstring = false, whitelist = [], categories, regions, severityThreshold = 0, indonesianVariation = false, detectSimilarity = false, similarityThreshold = 0.8, detectSplit = false, } = { ...DEFAULT_OPTIONS, ...options };
|
|
494
1135
|
const normalizedText = normalizeText(text);
|
|
495
|
-
let wordsToCheck = wordList;
|
|
1136
|
+
let wordsToCheck = wordList.length > 0 ? wordList : [];
|
|
496
1137
|
if (wordsToCheck.length === 0) {
|
|
497
1138
|
if (categories || regions || severityThreshold > 0) {
|
|
498
1139
|
wordsToCheck = wordObjects
|
|
@@ -516,99 +1157,68 @@ function findProfanity(text, options = {}) {
|
|
|
516
1157
|
}
|
|
517
1158
|
const matches = new Set();
|
|
518
1159
|
wordsToCheck.forEach((word) => {
|
|
519
|
-
const
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
1160
|
+
const regex = createWordRegex(word, {
|
|
1161
|
+
wholeWord: !checkSubstring,
|
|
1162
|
+
caseSensitive: false,
|
|
1163
|
+
leetSpeak: false,
|
|
1164
|
+
detectSplit: false,
|
|
1165
|
+
indonesianVariation: false,
|
|
1166
|
+
});
|
|
1167
|
+
while ((regex.exec(normalizedText)) !== null) {
|
|
1168
|
+
matches.add(word.toLowerCase());
|
|
524
1169
|
}
|
|
525
1170
|
});
|
|
526
|
-
// jika detectLeetSpeak diaktifkan, cari variasi leet speak
|
|
527
1171
|
if (detectLeetSpeak) {
|
|
528
1172
|
wordsToCheck.forEach((word) => {
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
1173
|
+
const leetRegex = createWordRegex(word, {
|
|
1174
|
+
wholeWord: !checkSubstring,
|
|
1175
|
+
caseSensitive: false,
|
|
1176
|
+
leetSpeak: true,
|
|
1177
|
+
detectSplit: false,
|
|
1178
|
+
indonesianVariation: false,
|
|
1179
|
+
});
|
|
532
1180
|
while ((leetRegex.exec(text)) !== null) {
|
|
533
1181
|
matches.add(word.toLowerCase());
|
|
534
1182
|
}
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
1183
|
+
});
|
|
1184
|
+
}
|
|
1185
|
+
if (indonesianVariation) {
|
|
1186
|
+
wordsToCheck.forEach((word) => {
|
|
1187
|
+
const variantRegex = createWordRegex(word, {
|
|
1188
|
+
wholeWord: !checkSubstring,
|
|
1189
|
+
caseSensitive: false,
|
|
1190
|
+
leetSpeak: false,
|
|
1191
|
+
detectSplit: false,
|
|
1192
|
+
indonesianVariation: true,
|
|
1193
|
+
});
|
|
1194
|
+
while ((variantRegex.exec(text)) !== null) {
|
|
539
1195
|
matches.add(word.toLowerCase());
|
|
540
1196
|
}
|
|
541
1197
|
});
|
|
542
1198
|
}
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
}
|
|
558
|
-
/**
|
|
559
|
-
* Escape karakter khusus dalam regex
|
|
560
|
-
*
|
|
561
|
-
* @param string String untuk di-escape
|
|
562
|
-
* @returns String yang sudah di-escape
|
|
563
|
-
*/
|
|
564
|
-
function escapeRegExp$2(string) {
|
|
565
|
-
return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
566
|
-
}
|
|
567
|
-
/**
|
|
568
|
-
* Pattern untuk leet speak
|
|
569
|
-
*
|
|
570
|
-
* @param word Kata untuk dibuat pattern leet speak
|
|
571
|
-
* @return Pattern regex untuk leet speak
|
|
572
|
-
*/
|
|
573
|
-
function createLeetSpeakPattern(word) {
|
|
574
|
-
const leetMap = {
|
|
575
|
-
a: ['a', '4', '@'],
|
|
576
|
-
b: ['b', '8', '6'],
|
|
577
|
-
c: ['c', '<', '(', '{'],
|
|
578
|
-
e: ['e', '3'],
|
|
579
|
-
g: ['g', '9'],
|
|
580
|
-
i: ['i', '1', '!'],
|
|
581
|
-
l: ['l', '1', '|'],
|
|
582
|
-
o: ['o', '0'],
|
|
583
|
-
s: ['s', '5', '$'],
|
|
584
|
-
t: ['t', '7', '+'],
|
|
585
|
-
z: ['z', '2'],
|
|
586
|
-
};
|
|
587
|
-
return word
|
|
588
|
-
.split('')
|
|
589
|
-
.map((char) => {
|
|
590
|
-
const lowerChar = char.toLowerCase();
|
|
591
|
-
const replacements = leetMap[lowerChar];
|
|
592
|
-
if (replacements && replacements.length > 0) {
|
|
593
|
-
return `[${replacements.join('')}]`;
|
|
594
|
-
}
|
|
595
|
-
else {
|
|
596
|
-
return escapeRegExp$2(char);
|
|
1199
|
+
if (detectSplit) {
|
|
1200
|
+
if (detectSplitWords(text, wordsToCheck)) {
|
|
1201
|
+
wordsToCheck.forEach((word) => {
|
|
1202
|
+
const splitRegex = createWordRegex(word, {
|
|
1203
|
+
wholeWord: false,
|
|
1204
|
+
caseSensitive: false,
|
|
1205
|
+
leetSpeak: false,
|
|
1206
|
+
detectSplit: true,
|
|
1207
|
+
indonesianVariation: false,
|
|
1208
|
+
});
|
|
1209
|
+
if (splitRegex.test(text)) {
|
|
1210
|
+
matches.add(word.toLowerCase());
|
|
1211
|
+
}
|
|
1212
|
+
});
|
|
597
1213
|
}
|
|
598
|
-
}
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
*/
|
|
607
|
-
function createEvasionPattern(word) {
|
|
608
|
-
return word
|
|
609
|
-
.split('')
|
|
610
|
-
.map((char) => escapeRegExp$2(char))
|
|
611
|
-
.join('[\\s\\-._*+]?');
|
|
1214
|
+
}
|
|
1215
|
+
if (detectSimilarity) {
|
|
1216
|
+
const possibleProfanity = findPossibleProfanityBySimiliarity(text, wordsToCheck, similarityThreshold);
|
|
1217
|
+
possibleProfanity.forEach((item) => {
|
|
1218
|
+
matches.add(item.original.toLowerCase());
|
|
1219
|
+
});
|
|
1220
|
+
}
|
|
1221
|
+
return Array.from(matches);
|
|
612
1222
|
}
|
|
613
1223
|
/**
|
|
614
1224
|
* Mencari kata kotor lengkap dengan metadata
|
|
@@ -634,8 +1244,8 @@ function findProfanityWithMetadata(text, options = {}) {
|
|
|
634
1244
|
/**
|
|
635
1245
|
* Mencari kategory kata kotor yang ada dalam teks
|
|
636
1246
|
*
|
|
637
|
-
* @param
|
|
638
|
-
* @
|
|
1247
|
+
* @param matchDetails Hasil pencarian dari fingProfanityWithMetadata()
|
|
1248
|
+
* @return Array kategori unik
|
|
639
1249
|
*/
|
|
640
1250
|
function findCategories(matchDetails) {
|
|
641
1251
|
const categories = new Set();
|
|
@@ -647,8 +1257,8 @@ function findCategories(matchDetails) {
|
|
|
647
1257
|
/**
|
|
648
1258
|
* Mencari region kata kotor yang ada dalam teks
|
|
649
1259
|
*
|
|
650
|
-
* @param
|
|
651
|
-
* @
|
|
1260
|
+
* @param matchDetails Hasil pencarian dari fingProfanityWithMetadata()
|
|
1261
|
+
* @return Array region unik
|
|
652
1262
|
*/
|
|
653
1263
|
function findRegions(matchDetails) {
|
|
654
1264
|
const regions = new Set();
|
|
@@ -697,12 +1307,13 @@ function calculateSeverity(matchDetails) {
|
|
|
697
1307
|
* @returns FilterResult dengan hasil filter
|
|
698
1308
|
*/
|
|
699
1309
|
function filter(text, options = {}) {
|
|
700
|
-
const { replaceWith =
|
|
1310
|
+
const { replaceWith = "*", fullWordCensor = true, detectLeetSpeak = true, whitelist = [], checkSubstring = false, useRandomGrawlix = false, keepFirstAndLast = false, indonesianVariation = false, } = { ...DEFAULT_OPTIONS, ...options };
|
|
701
1311
|
const matches = findProfanity(text, {
|
|
702
1312
|
...options,
|
|
703
1313
|
detectLeetSpeak,
|
|
704
1314
|
whitelist,
|
|
705
1315
|
checkSubstring,
|
|
1316
|
+
indonesianVariation,
|
|
706
1317
|
});
|
|
707
1318
|
const matchDetails = findProfanityWithMetadata(text, options);
|
|
708
1319
|
if (matches.length === 0) {
|
|
@@ -718,26 +1329,36 @@ function filter(text, options = {}) {
|
|
|
718
1329
|
const metadata = matchDetails.find((m) => m.word.toLowerCase() === word.toLowerCase() ||
|
|
719
1330
|
(m.aliases &&
|
|
720
1331
|
m.aliases.some((alias) => alias.toLowerCase() === word.toLowerCase())));
|
|
721
|
-
const regex =
|
|
1332
|
+
const regex = createWordRegex(word, {
|
|
1333
|
+
wholeWord: true,
|
|
1334
|
+
caseSensitive: false,
|
|
1335
|
+
leetSpeak: false,
|
|
1336
|
+
detectSplit: false,
|
|
1337
|
+
indonesianVariation: false,
|
|
1338
|
+
});
|
|
722
1339
|
let match;
|
|
723
|
-
|
|
1340
|
+
const textToSearch = filteredText;
|
|
1341
|
+
regex.lastIndex = 0;
|
|
1342
|
+
while ((match = regex.exec(textToSearch)) !== null) {
|
|
724
1343
|
const originalWord = match[0];
|
|
725
1344
|
if (whitelist.includes(originalWord.toLowerCase()))
|
|
726
1345
|
continue;
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
1346
|
+
let censoredWord;
|
|
1347
|
+
if (useRandomGrawlix) {
|
|
1348
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
1349
|
+
}
|
|
1350
|
+
else {
|
|
1351
|
+
censoredWord = censorWord(originalWord, replaceWith, !fullWordCensor && keepFirstAndLast);
|
|
1352
|
+
}
|
|
730
1353
|
replacements.push({
|
|
731
1354
|
original: originalWord,
|
|
732
1355
|
censored: censoredWord,
|
|
733
1356
|
metadata,
|
|
734
1357
|
});
|
|
735
|
-
const replaceRegex = new RegExp(`\\b${escapeRegExp
|
|
1358
|
+
const replaceRegex = new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g");
|
|
736
1359
|
filteredText = filteredText.replace(replaceRegex, censoredWord);
|
|
737
1360
|
}
|
|
738
1361
|
});
|
|
739
|
-
// if (detectLeetSpeak) {
|
|
740
|
-
// }
|
|
741
1362
|
return {
|
|
742
1363
|
filtered: filteredText,
|
|
743
1364
|
censored: replacements.length,
|
|
@@ -755,28 +1376,6 @@ function isProfane(text, options = {}) {
|
|
|
755
1376
|
const matches = findProfanity(text, options);
|
|
756
1377
|
return matches.length > 0;
|
|
757
1378
|
}
|
|
758
|
-
/**
|
|
759
|
-
* Escape karakter khusus regex
|
|
760
|
-
*
|
|
761
|
-
* @param string String untuk di-escape
|
|
762
|
-
* @returns String yang telah di-escape
|
|
763
|
-
*/
|
|
764
|
-
function escapeRegExp$1(string) {
|
|
765
|
-
return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
766
|
-
}
|
|
767
|
-
/**
|
|
768
|
-
* Menyensor sebagian kata
|
|
769
|
-
* @param word Kata yang akan disensor
|
|
770
|
-
* @param replaceChar Karakter pengganti
|
|
771
|
-
* @returns Kata yang sudah disensor sebagian
|
|
772
|
-
*/
|
|
773
|
-
function censorPartialWord(word, replaceChar) {
|
|
774
|
-
if (word.length <= 2) {
|
|
775
|
-
return replaceChar.repeat(word.length);
|
|
776
|
-
}
|
|
777
|
-
// Simpan huruf pertama dan terakhir, sensor yang lain
|
|
778
|
-
return word[0] + replaceChar.repeat(word.length - 2) + word[word.length - 1];
|
|
779
|
-
}
|
|
780
1379
|
|
|
781
1380
|
/**
|
|
782
1381
|
* Menganalisis teks untuk kata kotor
|
|
@@ -786,7 +1385,8 @@ function censorPartialWord(word, replaceChar) {
|
|
|
786
1385
|
* @return AnalysisResult dengan hasil analisis
|
|
787
1386
|
*/
|
|
788
1387
|
function analyze(text, options = {}) {
|
|
789
|
-
const
|
|
1388
|
+
const mergedOptions = { ...DEFAULT_OPTIONS, ...options };
|
|
1389
|
+
const matches = findProfanity(text, mergedOptions);
|
|
790
1390
|
if (matches.length === 0) {
|
|
791
1391
|
return {
|
|
792
1392
|
hasProfanity: false,
|
|
@@ -797,10 +1397,15 @@ function analyze(text, options = {}) {
|
|
|
797
1397
|
severityScore: 0,
|
|
798
1398
|
};
|
|
799
1399
|
}
|
|
800
|
-
const matchDetails = findProfanityWithMetadata(text,
|
|
1400
|
+
const matchDetails = findProfanityWithMetadata(text, mergedOptions);
|
|
801
1401
|
const categories = findCategories(matchDetails);
|
|
802
1402
|
const regions = findRegions(matchDetails);
|
|
803
1403
|
const severityScore = calculateSeverity(matchDetails);
|
|
1404
|
+
let similarWords = [];
|
|
1405
|
+
if (mergedOptions.detectSimilarity) {
|
|
1406
|
+
const wordList = matchDetails.map((word) => word.word);
|
|
1407
|
+
similarWords = findPossibleProfanityBySimiliarity(text, wordList, mergedOptions.similarityThreshold || 0.8);
|
|
1408
|
+
}
|
|
804
1409
|
return {
|
|
805
1410
|
hasProfanity: true,
|
|
806
1411
|
matches,
|
|
@@ -808,17 +1413,19 @@ function analyze(text, options = {}) {
|
|
|
808
1413
|
categories,
|
|
809
1414
|
regions,
|
|
810
1415
|
severityScore,
|
|
1416
|
+
similarWords: mergedOptions.detectSimilarity ? similarWords : undefined,
|
|
811
1417
|
};
|
|
812
1418
|
}
|
|
813
1419
|
/**
|
|
814
|
-
*
|
|
1420
|
+
* Menganalisis daftar teks dan ringkasan
|
|
815
1421
|
*
|
|
816
|
-
* @param texts Daftar teks
|
|
1422
|
+
* @param texts Daftar teks yang dianalisis
|
|
817
1423
|
* @param options Opsi untuk analisis
|
|
818
1424
|
* @return Objek dengan ringkasan analisis
|
|
819
1425
|
*/
|
|
820
1426
|
function batchAnalyze(texts, options = {}) {
|
|
821
|
-
const
|
|
1427
|
+
const mergedOptions = { ...DEFAULT_OPTIONS, ...options };
|
|
1428
|
+
const results = texts.map((text) => analyze(text, mergedOptions));
|
|
822
1429
|
const profaneTexts = results.filter((result) => result.hasProfanity).length;
|
|
823
1430
|
const totalSeverity = results.reduce((sum, result) => sum + result.severityScore, 0);
|
|
824
1431
|
const averageSeverity = profaneTexts > 0 ? totalSeverity / profaneTexts : 0;
|
|
@@ -865,11 +1472,10 @@ function batchAnalyze(texts, options = {}) {
|
|
|
865
1472
|
* @returns Array hasil analisis per-kalimat
|
|
866
1473
|
*/
|
|
867
1474
|
function analyzeBySentence(text, options = {}) {
|
|
868
|
-
const sentences = text
|
|
869
|
-
|
|
870
|
-
.filter((sentence) => sentence.trim().length > 0);
|
|
1475
|
+
const sentences = splitIntoSentences(text);
|
|
1476
|
+
const mergedOptions = { ...DEFAULT_OPTIONS, ...options };
|
|
871
1477
|
return sentences.map((sentence) => {
|
|
872
|
-
const result = analyze(sentence,
|
|
1478
|
+
const result = analyze(sentence, mergedOptions);
|
|
873
1479
|
return {
|
|
874
1480
|
...result,
|
|
875
1481
|
sentence,
|
|
@@ -880,23 +1486,24 @@ function analyzeBySentence(text, options = {}) {
|
|
|
880
1486
|
* Menganalisis teks untuk menemukan kata kotor pada konteks tertentu
|
|
881
1487
|
*
|
|
882
1488
|
* @param text Teks yang akan dianalisis
|
|
883
|
-
* @param
|
|
1489
|
+
* @param contextWindowSize Ukuran konteks (jumlah kata) di sekitar kata kotor
|
|
884
1490
|
* @param options Opsi untuk analisis
|
|
885
1491
|
* @returns Konteks di dekat kata kotor
|
|
886
1492
|
*/
|
|
887
1493
|
function analyzeWithContext(text, contextWindowSize = 5, options = {}) {
|
|
888
|
-
const
|
|
1494
|
+
const mergedOptions = { ...DEFAULT_OPTIONS, ...options };
|
|
1495
|
+
const matches = findProfanity(text, mergedOptions);
|
|
889
1496
|
if (matches.length === 0) {
|
|
890
1497
|
return [];
|
|
891
1498
|
}
|
|
892
1499
|
const result = [];
|
|
893
1500
|
for (const word of matches) {
|
|
894
|
-
const regex =
|
|
1501
|
+
const regex = createContextRegex(word, contextWindowSize);
|
|
895
1502
|
let match;
|
|
896
1503
|
while ((match = regex.exec(text)) !== null) {
|
|
897
|
-
const beforeContext = match[1] ||
|
|
1504
|
+
const beforeContext = match[1] || "";
|
|
898
1505
|
const wordMatch = match[2];
|
|
899
|
-
const afterContext = match[3] ||
|
|
1506
|
+
const afterContext = match[3] || "";
|
|
900
1507
|
result.push({
|
|
901
1508
|
word: wordMatch,
|
|
902
1509
|
context: beforeContext + wordMatch + afterContext,
|
|
@@ -909,9 +1516,6 @@ function analyzeWithContext(text, contextWindowSize = 5, options = {}) {
|
|
|
909
1516
|
}
|
|
910
1517
|
return result;
|
|
911
1518
|
}
|
|
912
|
-
function escapeRegExp(string) {
|
|
913
|
-
return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
914
|
-
}
|
|
915
1519
|
|
|
916
1520
|
class IDProfanityFilter {
|
|
917
1521
|
/**
|
|
@@ -919,7 +1523,7 @@ class IDProfanityFilter {
|
|
|
919
1523
|
* @param options Opsi untuk filter
|
|
920
1524
|
*/
|
|
921
1525
|
constructor(options = {}) {
|
|
922
|
-
this.options = options;
|
|
1526
|
+
this.options = { ...DEFAULT_OPTIONS, ...options };
|
|
923
1527
|
}
|
|
924
1528
|
/**
|
|
925
1529
|
* Menyensor teks yang diberikan
|
|
@@ -980,6 +1584,14 @@ class IDProfanityFilter {
|
|
|
980
1584
|
...options,
|
|
981
1585
|
};
|
|
982
1586
|
}
|
|
1587
|
+
/**
|
|
1588
|
+
* Menggunakan preset yang telah ditentukan
|
|
1589
|
+
* @param presetName Nama preset yang akan digunakan
|
|
1590
|
+
* @param additionalOptions Opsi tambahan untuk override
|
|
1591
|
+
*/
|
|
1592
|
+
usePreset(presetName, additionalOptions = {}) {
|
|
1593
|
+
this.options = getPresetOptions(presetName, additionalOptions);
|
|
1594
|
+
}
|
|
983
1595
|
/**
|
|
984
1596
|
* Menetapkan daftar kata kustom
|
|
985
1597
|
* @param wordList Daftar kata untuk digunakan
|
|
@@ -1006,26 +1618,86 @@ class IDProfanityFilter {
|
|
|
1006
1618
|
return;
|
|
1007
1619
|
this.options.whitelist = this.options.whitelist.filter((w) => w.toLowerCase() !== word.toLowerCase());
|
|
1008
1620
|
}
|
|
1621
|
+
/**
|
|
1622
|
+
* Mengaktifkan deteksi variasi ejaan Indonesia
|
|
1623
|
+
*/
|
|
1624
|
+
enableIndonesianVariations() {
|
|
1625
|
+
this.options.indonesianVariation = true;
|
|
1626
|
+
}
|
|
1627
|
+
/**
|
|
1628
|
+
* Mengaktifkan deteksi kata yang dipisah
|
|
1629
|
+
*/
|
|
1630
|
+
enableSplitWordDetection() {
|
|
1631
|
+
this.options.detectSplit = true;
|
|
1632
|
+
}
|
|
1633
|
+
/**
|
|
1634
|
+
* Mengaktifkan deteksi berdasarkan kesamaan
|
|
1635
|
+
* @param threshold Threshold kesamaan (0-1)
|
|
1636
|
+
*/
|
|
1637
|
+
enableSimilarityDetection(threshold = 0.8) {
|
|
1638
|
+
this.options.detectSimilarity = true;
|
|
1639
|
+
this.options.similarityThreshold = threshold;
|
|
1640
|
+
}
|
|
1009
1641
|
}
|
|
1010
1642
|
const idFilter = {
|
|
1011
|
-
filter: (text, options) => filter(text, options),
|
|
1012
|
-
isProfane: (text, options) => isProfane(text, options),
|
|
1013
|
-
analyze: (text, options) => analyze(text, options),
|
|
1014
|
-
batchAnalyze: (texts, options) => batchAnalyze(texts, options),
|
|
1643
|
+
filter: (text, options) => filter(text, { ...DEFAULT_OPTIONS, ...options }),
|
|
1644
|
+
isProfane: (text, options) => isProfane(text, { ...DEFAULT_OPTIONS, ...options }),
|
|
1645
|
+
analyze: (text, options) => analyze(text, { ...DEFAULT_OPTIONS, ...options }),
|
|
1646
|
+
batchAnalyze: (texts, options) => batchAnalyze(texts, { ...DEFAULT_OPTIONS, ...options }),
|
|
1647
|
+
getPresetOptions,
|
|
1648
|
+
presets: {
|
|
1649
|
+
filter: FILTER_PRESETS,
|
|
1650
|
+
category: CATEGORY_PRESETS,
|
|
1651
|
+
region: REGION_PRESETS,
|
|
1652
|
+
},
|
|
1015
1653
|
};
|
|
1016
1654
|
|
|
1655
|
+
exports.CATEGORY_PRESETS = CATEGORY_PRESETS;
|
|
1656
|
+
exports.DEFAULT_OPTIONS = DEFAULT_OPTIONS;
|
|
1657
|
+
exports.FILTER_PRESETS = FILTER_PRESETS;
|
|
1017
1658
|
exports.IDProfanityFilter = IDProfanityFilter;
|
|
1659
|
+
exports.REGION_PRESETS = REGION_PRESETS;
|
|
1660
|
+
exports.REPLACEMENT_CHARS = REPLACEMENT_CHARS;
|
|
1661
|
+
exports.addIndonesianVariations = addIndonesianVariations;
|
|
1662
|
+
exports.addLeetSpeakVariations = addLeetSpeakVariations;
|
|
1663
|
+
exports.addSplitVariations = addSplitVariations;
|
|
1018
1664
|
exports.analyze = analyze;
|
|
1019
1665
|
exports.analyzeBySentence = analyzeBySentence;
|
|
1020
1666
|
exports.analyzeWithContext = analyzeWithContext;
|
|
1021
1667
|
exports.batchAnalyze = batchAnalyze;
|
|
1022
1668
|
exports.calculateSeverity = calculateSeverity;
|
|
1669
|
+
exports.censorWord = censorWord;
|
|
1670
|
+
exports.clusterSimilarWords = clusterSimilarWords;
|
|
1671
|
+
exports.containsAnyWord = containsAnyWord;
|
|
1672
|
+
exports.containsEuphemism = containsEuphemism;
|
|
1673
|
+
exports.createContextRegex = createContextRegex;
|
|
1674
|
+
exports.createEvasionRegex = createEvasionRegex;
|
|
1675
|
+
exports.createIndonesianVariationRegex = createIndonesianVariationRegex;
|
|
1676
|
+
exports.createOptions = createOptions;
|
|
1677
|
+
exports.createWordFormRegex = createWordFormRegex;
|
|
1678
|
+
exports.createWordRegex = createWordRegex;
|
|
1023
1679
|
exports.default = IDProfanityFilter;
|
|
1680
|
+
exports.detectSplitWords = detectSplitWords;
|
|
1681
|
+
exports.escapeRegExp = escapeRegExp;
|
|
1024
1682
|
exports.filter = filter;
|
|
1025
1683
|
exports.findCategories = findCategories;
|
|
1684
|
+
exports.findMostSimilar = findMostSimilar;
|
|
1685
|
+
exports.findPossibleProfanityBySimiliarity = findPossibleProfanityBySimiliarity;
|
|
1026
1686
|
exports.findProfanity = findProfanity;
|
|
1027
1687
|
exports.findProfanityWithMetadata = findProfanityWithMetadata;
|
|
1028
1688
|
exports.findRegions = findRegions;
|
|
1689
|
+
exports.getContextAroundIndex = getContextAroundIndex;
|
|
1690
|
+
exports.getPresetOptions = getPresetOptions;
|
|
1691
|
+
exports.getRandomGrawlix = getRandomGrawlix;
|
|
1692
|
+
exports.getReplacementChar = getReplacementChar;
|
|
1029
1693
|
exports.idFilter = idFilter;
|
|
1694
|
+
exports.isPossibleProfanityVariation = isPossibleProfanityVariation;
|
|
1030
1695
|
exports.isProfane = isProfane;
|
|
1696
|
+
exports.levenshteinDistance = levenshteinDistance;
|
|
1697
|
+
exports.makeRandomGrawlixString = makeRandomGrawlixString;
|
|
1698
|
+
exports.maskText = maskText;
|
|
1699
|
+
exports.normalizeText = normalizeText;
|
|
1700
|
+
exports.splitIntoSentences = splitIntoSentences;
|
|
1701
|
+
exports.stringSimilarity = stringSimilarity;
|
|
1702
|
+
exports.toLeetSpeak = toLeetSpeak;
|
|
1031
1703
|
//# sourceMappingURL=index.js.map
|