@sideid/id-profanity-filter 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.eslintrc.js +16 -0
- package/.github/workflows/ci.yml +0 -0
- package/CONTRIBUTING.md +151 -0
- package/LICENSE +21 -0
- package/README.md +285 -0
- package/dist/constants/categories/insult.d.ts +4 -0
- package/dist/constants/categories/sexual.d.ts +4 -0
- package/dist/constants/regions/batak.d.ts +4 -0
- package/dist/constants/regions/betawi.d.ts +4 -0
- package/dist/constants/regions/general.d.ts +4 -0
- package/dist/constants/regions/jawa.d.ts +4 -0
- package/dist/constants/regions/sunda.d.ts +4 -0
- package/dist/constants/wordList.d.ts +34 -0
- package/dist/core/analyzer.d.ts +54 -0
- package/dist/core/filter.d.ts +17 -0
- package/dist/core/matcher.d.ts +31 -0
- package/dist/index.d.ts +71 -0
- package/dist/index.esm.js +1014 -0
- package/dist/index.esm.js.map +1 -0
- package/dist/index.js +1031 -0
- package/dist/index.js.map +1 -0
- package/dist/types/index.d.ts +69 -0
- package/examples/basic.ts +52 -0
- package/jest.config.js +10 -0
- package/package.json +51 -0
- package/prettierrc +7 -0
- package/rollup.config.js +35 -0
- package/src/config/options.ts +0 -0
- package/src/constants/categories/blasphemy.ts +0 -0
- package/src/constants/categories/disgusting.ts +0 -0
- package/src/constants/categories/drugs.ts +0 -0
- package/src/constants/categories/index.ts +31 -0
- package/src/constants/categories/insult.ts +114 -0
- package/src/constants/categories/profanity.ts +0 -0
- package/src/constants/categories/sexual.ts +100 -0
- package/src/constants/categories/slur.ts +0 -0
- package/src/constants/regions/aceh.ts +0 -0
- package/src/constants/regions/ambon.ts +0 -0
- package/src/constants/regions/bali.ts +0 -0
- package/src/constants/regions/banjar.ts +0 -0
- package/src/constants/regions/batak.ts +17 -0
- package/src/constants/regions/betawi.ts +39 -0
- package/src/constants/regions/bugis.ts +0 -0
- package/src/constants/regions/general.ts +111 -0
- package/src/constants/regions/index.ts +62 -0
- package/src/constants/regions/jawa.ts +102 -0
- package/src/constants/regions/lampung.ts +0 -0
- package/src/constants/regions/madura.ts +0 -0
- package/src/constants/regions/manado.ts +0 -0
- package/src/constants/regions/minang.ts +0 -0
- package/src/constants/regions/ntb.ts +0 -0
- package/src/constants/regions/ntt.ts +0 -0
- package/src/constants/regions/palembang.ts +0 -0
- package/src/constants/regions/papua.ts +0 -0
- package/src/constants/regions/sunda.ts +35 -0
- package/src/constants/wordList.ts +125 -0
- package/src/core/analyzer.ts +204 -0
- package/src/core/filter.ts +129 -0
- package/src/core/matcher.ts +267 -0
- package/src/index.ts +133 -0
- package/src/types/index.ts +115 -0
- package/src/utils/regexUtils.ts +0 -0
- package/src/utils/stringUtils.ts +0 -0
- package/tsconfig.json +115 -0
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
import { FilterOptions, FilterResult, ProfanityWord } from '../types';
|
|
2
|
+
import { findProfanity, findProfanityWithMetadata } from './matcher';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Menyensor kata kotor dalam teks
|
|
6
|
+
*
|
|
7
|
+
* @param text Teks yang akan disensor
|
|
8
|
+
* @param options Opsi untuk filter
|
|
9
|
+
* @returns FilterResult dengan hasil filter
|
|
10
|
+
*/
|
|
11
|
+
export function filter(
|
|
12
|
+
text: string,
|
|
13
|
+
options: FilterOptions = {},
|
|
14
|
+
): FilterResult {
|
|
15
|
+
const {
|
|
16
|
+
replaceWith = '*',
|
|
17
|
+
fullWordCensor = true,
|
|
18
|
+
detectLeetSpeak = true,
|
|
19
|
+
whitelist = [],
|
|
20
|
+
checkSubstring = false,
|
|
21
|
+
} = options;
|
|
22
|
+
|
|
23
|
+
const matches = findProfanity(text, {
|
|
24
|
+
...options,
|
|
25
|
+
detectLeetSpeak,
|
|
26
|
+
whitelist,
|
|
27
|
+
checkSubstring,
|
|
28
|
+
});
|
|
29
|
+
|
|
30
|
+
const matchDetails = findProfanityWithMetadata(text, options);
|
|
31
|
+
|
|
32
|
+
if (matches.length === 0) {
|
|
33
|
+
return {
|
|
34
|
+
filtered: text,
|
|
35
|
+
censored: 0,
|
|
36
|
+
replacements: [],
|
|
37
|
+
};
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
let filteredText = text;
|
|
41
|
+
|
|
42
|
+
const replacements: Array<{
|
|
43
|
+
original: string;
|
|
44
|
+
censored: string;
|
|
45
|
+
metadata?: ProfanityWord;
|
|
46
|
+
}> = [];
|
|
47
|
+
|
|
48
|
+
matches.forEach((word) => {
|
|
49
|
+
const metadata = matchDetails.find(
|
|
50
|
+
(m) =>
|
|
51
|
+
m.word.toLowerCase() === word.toLowerCase() ||
|
|
52
|
+
(m.aliases &&
|
|
53
|
+
m.aliases.some(
|
|
54
|
+
(alias) => alias.toLowerCase() === word.toLowerCase(),
|
|
55
|
+
)),
|
|
56
|
+
);
|
|
57
|
+
|
|
58
|
+
const regex = new RegExp(`\\b${escapeRegExp(word)}\\b`, 'gi');
|
|
59
|
+
|
|
60
|
+
let match;
|
|
61
|
+
while ((match = regex.exec(filteredText)) !== null) {
|
|
62
|
+
const originalWord = match[0];
|
|
63
|
+
|
|
64
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
65
|
+
|
|
66
|
+
const censoredWord = fullWordCensor
|
|
67
|
+
? replaceWith.repeat(originalWord.length)
|
|
68
|
+
: censorPartialWord(originalWord, replaceWith);
|
|
69
|
+
|
|
70
|
+
replacements.push({
|
|
71
|
+
original: originalWord,
|
|
72
|
+
censored: censoredWord,
|
|
73
|
+
metadata,
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
const replaceRegex = new RegExp(
|
|
77
|
+
`\\b${escapeRegExp(originalWord)}\\b`,
|
|
78
|
+
'g',
|
|
79
|
+
);
|
|
80
|
+
filteredText = filteredText.replace(replaceRegex, censoredWord);
|
|
81
|
+
}
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
// if (detectLeetSpeak) {
|
|
85
|
+
// }
|
|
86
|
+
|
|
87
|
+
return {
|
|
88
|
+
filtered: filteredText,
|
|
89
|
+
censored: replacements.length,
|
|
90
|
+
replacements,
|
|
91
|
+
};
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Memeriksa apakah teks mengandung kata kotor
|
|
96
|
+
*
|
|
97
|
+
* @param text Teks yang akan diperiksa
|
|
98
|
+
* @param options Opsi untuk pemeriksaan
|
|
99
|
+
* @returns Boolean apakah teks mengandung kata kotor
|
|
100
|
+
*/
|
|
101
|
+
export function isProfane(text: string, options: FilterOptions = {}): boolean {
|
|
102
|
+
const matches = findProfanity(text, options);
|
|
103
|
+
return matches.length > 0;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Escape karakter khusus regex
|
|
108
|
+
*
|
|
109
|
+
* @param string String untuk di-escape
|
|
110
|
+
* @returns String yang telah di-escape
|
|
111
|
+
*/
|
|
112
|
+
function escapeRegExp(string: string): string {
|
|
113
|
+
return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Menyensor sebagian kata
|
|
118
|
+
* @param word Kata yang akan disensor
|
|
119
|
+
* @param replaceChar Karakter pengganti
|
|
120
|
+
* @returns Kata yang sudah disensor sebagian
|
|
121
|
+
*/
|
|
122
|
+
function censorPartialWord(word: string, replaceChar: string): string {
|
|
123
|
+
if (word.length <= 2) {
|
|
124
|
+
return replaceChar.repeat(word.length);
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
// Simpan huruf pertama dan terakhir, sensor yang lain
|
|
128
|
+
return word[0] + replaceChar.repeat(word.length - 2) + word[word.length - 1];
|
|
129
|
+
}
|
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
import {
|
|
2
|
+
ProfanityWord,
|
|
3
|
+
ProfanityCategory,
|
|
4
|
+
Region,
|
|
5
|
+
FilterOptions,
|
|
6
|
+
} from '../types';
|
|
7
|
+
|
|
8
|
+
import { wordObjects, getWordsByFilter } from '../constants/wordList';
|
|
9
|
+
|
|
10
|
+
export function findProfanity(
|
|
11
|
+
text: string,
|
|
12
|
+
options: FilterOptions = {},
|
|
13
|
+
): string[] {
|
|
14
|
+
const {
|
|
15
|
+
wordList = [],
|
|
16
|
+
detectLeetSpeak = true,
|
|
17
|
+
checkSubstring = false,
|
|
18
|
+
whitelist = [],
|
|
19
|
+
categories,
|
|
20
|
+
regions,
|
|
21
|
+
severityThreshold = 0,
|
|
22
|
+
} = options;
|
|
23
|
+
|
|
24
|
+
const normalizedText = normalizeText(text);
|
|
25
|
+
|
|
26
|
+
let wordsToCheck: string[] = wordList;
|
|
27
|
+
|
|
28
|
+
if (wordsToCheck.length === 0) {
|
|
29
|
+
if (categories || regions || severityThreshold > 0) {
|
|
30
|
+
wordsToCheck = wordObjects
|
|
31
|
+
.filter((word) => {
|
|
32
|
+
const matchCategory = categories
|
|
33
|
+
? categories.includes(word.category)
|
|
34
|
+
: true;
|
|
35
|
+
const matchRegion = regions ? regions.includes(word.region) : true;
|
|
36
|
+
const matchSeverity = word.severity >= severityThreshold;
|
|
37
|
+
return matchCategory && matchRegion && matchSeverity;
|
|
38
|
+
})
|
|
39
|
+
.map((word) => word.word);
|
|
40
|
+
} else {
|
|
41
|
+
wordsToCheck = wordObjects.map((word) => word.word);
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
wordsToCheck = wordsToCheck.filter(
|
|
46
|
+
(word) => !whitelist.includes(word.toLocaleLowerCase()),
|
|
47
|
+
);
|
|
48
|
+
|
|
49
|
+
if (wordsToCheck.length === 0) {
|
|
50
|
+
return [];
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
const matches = new Set<string>();
|
|
54
|
+
|
|
55
|
+
wordsToCheck.forEach((word) => {
|
|
56
|
+
const pattern = checkSubstring ? word : `\\b${escapeRegExp(word)}\\b`;
|
|
57
|
+
|
|
58
|
+
const regex = new RegExp(pattern, 'gi');
|
|
59
|
+
|
|
60
|
+
let match;
|
|
61
|
+
while ((match = regex.exec(normalizedText)) !== null) {
|
|
62
|
+
matches.add(match[0].toLowerCase());
|
|
63
|
+
}
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
// jika detectLeetSpeak diaktifkan, cari variasi leet speak
|
|
67
|
+
if (detectLeetSpeak) {
|
|
68
|
+
wordsToCheck.forEach((word) => {
|
|
69
|
+
// Cek leet speak
|
|
70
|
+
const leetPattern = createLeetSpeakPattern(word);
|
|
71
|
+
const leetRegex = new RegExp(
|
|
72
|
+
checkSubstring ? leetPattern : `\\b${leetPattern}\\b`,
|
|
73
|
+
'gi',
|
|
74
|
+
);
|
|
75
|
+
|
|
76
|
+
let match;
|
|
77
|
+
while ((match = leetRegex.exec(text)) !== null) {
|
|
78
|
+
matches.add(word.toLowerCase());
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
// cek karakter yang dipisah
|
|
82
|
+
const evasionPattern = createEvasionPattern(word);
|
|
83
|
+
const evasionRegex = new RegExp(evasionPattern, 'gi');
|
|
84
|
+
while ((match = evasionRegex.exec(text)) !== null) {
|
|
85
|
+
matches.add(word.toLowerCase());
|
|
86
|
+
}
|
|
87
|
+
});
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
return Array.from(matches);
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* Menormalisasi teks untuk perbandingan
|
|
95
|
+
*
|
|
96
|
+
* @param text Teks untuk dinormalisasi
|
|
97
|
+
* @returns Teks yang dinormalisasi
|
|
98
|
+
*/
|
|
99
|
+
function normalizeText(text: string): string {
|
|
100
|
+
return text
|
|
101
|
+
.toLowerCase()
|
|
102
|
+
.normalize('NFD') // Normalisasi Unicode
|
|
103
|
+
.replace(/[\u0300-\u036f]/g, '') // Hapus diacritic marks
|
|
104
|
+
.trim();
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Escape karakter khusus dalam regex
|
|
109
|
+
*
|
|
110
|
+
* @param string String untuk di-escape
|
|
111
|
+
* @returns String yang sudah di-escape
|
|
112
|
+
*/
|
|
113
|
+
function escapeRegExp(string: string): string {
|
|
114
|
+
return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Pattern untuk leet speak
|
|
119
|
+
*
|
|
120
|
+
* @param word Kata untuk dibuat pattern leet speak
|
|
121
|
+
* @return Pattern regex untuk leet speak
|
|
122
|
+
*/
|
|
123
|
+
function createLeetSpeakPattern(word: string): string {
|
|
124
|
+
const leetMap: Record<string, string[]> = {
|
|
125
|
+
a: ['a', '4', '@'],
|
|
126
|
+
b: ['b', '8', '6'],
|
|
127
|
+
c: ['c', '<', '(', '{'],
|
|
128
|
+
e: ['e', '3'],
|
|
129
|
+
g: ['g', '9'],
|
|
130
|
+
i: ['i', '1', '!'],
|
|
131
|
+
l: ['l', '1', '|'],
|
|
132
|
+
o: ['o', '0'],
|
|
133
|
+
s: ['s', '5', '$'],
|
|
134
|
+
t: ['t', '7', '+'],
|
|
135
|
+
z: ['z', '2'],
|
|
136
|
+
};
|
|
137
|
+
|
|
138
|
+
return word
|
|
139
|
+
.split('')
|
|
140
|
+
.map((char) => {
|
|
141
|
+
const lowerChar = char.toLowerCase();
|
|
142
|
+
const replacements = leetMap[lowerChar];
|
|
143
|
+
|
|
144
|
+
if (replacements && replacements.length > 0) {
|
|
145
|
+
return `[${replacements.join('')}]`;
|
|
146
|
+
} else {
|
|
147
|
+
return escapeRegExp(char);
|
|
148
|
+
}
|
|
149
|
+
})
|
|
150
|
+
.join('');
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/**
|
|
154
|
+
* Pattern untuk deteksi upaya menghindari filter
|
|
155
|
+
*
|
|
156
|
+
* @param word Kata untuk dibuat pattern evasion
|
|
157
|
+
* @return Pattern regex untuk evasion
|
|
158
|
+
*/
|
|
159
|
+
function createEvasionPattern(word: string): string {
|
|
160
|
+
return word
|
|
161
|
+
.split('')
|
|
162
|
+
.map((char) => escapeRegExp(char))
|
|
163
|
+
.join('[\\s\\-._*+]?');
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Mencari kata kotor lengkap dengan metadata
|
|
168
|
+
*
|
|
169
|
+
* @param text Teks yang akan diperika
|
|
170
|
+
* @param options Opsi utnuk pencarian kata kotor
|
|
171
|
+
* @return Array dari objek kata kotor yang ditemukan
|
|
172
|
+
*/
|
|
173
|
+
export function findProfanityWithMetadata(
|
|
174
|
+
text: string,
|
|
175
|
+
options: FilterOptions = {},
|
|
176
|
+
): ProfanityWord[] {
|
|
177
|
+
const matches = findProfanity(text, options);
|
|
178
|
+
if (matches.length === 0) {
|
|
179
|
+
return [];
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
return matches
|
|
183
|
+
.map((word) => {
|
|
184
|
+
const wordObject = wordObjects.find(
|
|
185
|
+
(obj) =>
|
|
186
|
+
obj.word.toLowerCase() === word.toLowerCase() ||
|
|
187
|
+
(obj.aliases &&
|
|
188
|
+
obj.aliases.some(
|
|
189
|
+
(alias) => alias.toLowerCase() === word.toLowerCase(),
|
|
190
|
+
)),
|
|
191
|
+
);
|
|
192
|
+
|
|
193
|
+
return wordObject;
|
|
194
|
+
})
|
|
195
|
+
.filter((word): word is ProfanityWord => word !== undefined);
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Mencari kategory kata kotor yang ada dalam teks
|
|
200
|
+
*
|
|
201
|
+
* @param text matchDetails Hasil pencarian dari fingProfanityWithMetadata()
|
|
202
|
+
* @param options Opsi untuk pencarian kategori
|
|
203
|
+
*/
|
|
204
|
+
export function findCategories(
|
|
205
|
+
matchDetails: ProfanityWord[],
|
|
206
|
+
): ProfanityCategory[] {
|
|
207
|
+
const categories = new Set<ProfanityCategory>();
|
|
208
|
+
|
|
209
|
+
matchDetails.forEach((word) => {
|
|
210
|
+
categories.add(word.category);
|
|
211
|
+
});
|
|
212
|
+
|
|
213
|
+
return Array.from(categories);
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
/**
|
|
217
|
+
* Mencari region kata kotor yang ada dalam teks
|
|
218
|
+
*
|
|
219
|
+
* @param text matchDetails Hasil pencarian dari fingProfanityWithMetadata()
|
|
220
|
+
* @param options Opsi untuk pencarian region
|
|
221
|
+
*/
|
|
222
|
+
export function findRegions(matchDetails: ProfanityWord[]): Region[] {
|
|
223
|
+
const regions = new Set<Region>();
|
|
224
|
+
|
|
225
|
+
matchDetails.forEach((word) => {
|
|
226
|
+
regions.add(word.region);
|
|
227
|
+
});
|
|
228
|
+
|
|
229
|
+
return Array.from(regions);
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* Menghitung tingkat keparahan kata kotor yang ditemukan
|
|
234
|
+
*
|
|
235
|
+
* @param matchDetails Hasil pencarian dari findProfanityWithMetadata()
|
|
236
|
+
* @return Skor keparahan dari 0-1
|
|
237
|
+
*/
|
|
238
|
+
export function calculateSeverity(matchDetails: ProfanityWord[]): number {
|
|
239
|
+
if (matchDetails.length === 0) {
|
|
240
|
+
return 0;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
const countFactor = Math.min(matchDetails.length / 10, 1); // Maksimal 10 kata
|
|
244
|
+
|
|
245
|
+
const categoryWeights: Record<ProfanityCategory, number> = {
|
|
246
|
+
sexual: 0.9,
|
|
247
|
+
blasphemy: 0.9,
|
|
248
|
+
slur: 0.8,
|
|
249
|
+
profanity: 0.7,
|
|
250
|
+
insult: 0.6,
|
|
251
|
+
drugs: 0.5,
|
|
252
|
+
disgusting: 0.5,
|
|
253
|
+
};
|
|
254
|
+
|
|
255
|
+
let severitySum = 0;
|
|
256
|
+
matchDetails.forEach((word) => {
|
|
257
|
+
const categoryWeight = categoryWeights[word.category] || 0.5;
|
|
258
|
+
const wordSeverity = word.severity * categoryWeight;
|
|
259
|
+
severitySum += wordSeverity;
|
|
260
|
+
});
|
|
261
|
+
|
|
262
|
+
const severityAvg = severitySum / matchDetails.length;
|
|
263
|
+
|
|
264
|
+
// Gabungkan jumlah kata dan keparahan rata-rata
|
|
265
|
+
// 70% keparahan kata + 30% faktor jumlah
|
|
266
|
+
return 0.7 * severityAvg + 0.3 * countFactor;
|
|
267
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
export * from './types';
|
|
2
|
+
export * from './core/matcher';
|
|
3
|
+
export * from './core/filter';
|
|
4
|
+
export * from './core/analyzer';
|
|
5
|
+
|
|
6
|
+
import { filter, isProfane } from './core/filter';
|
|
7
|
+
import {
|
|
8
|
+
analyze,
|
|
9
|
+
batchAnalyze,
|
|
10
|
+
analyzeBySentence,
|
|
11
|
+
analyzeWithContext,
|
|
12
|
+
} from './core/analyzer';
|
|
13
|
+
import { FilterOptions, FilterResult, AnalysisResult } from './types';
|
|
14
|
+
|
|
15
|
+
export class IDProfanityFilter {
|
|
16
|
+
private options: FilterOptions;
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Membuat instance filter baru
|
|
20
|
+
* @param options Opsi untuk filter
|
|
21
|
+
*/
|
|
22
|
+
constructor(options: FilterOptions = {}) {
|
|
23
|
+
this.options = options;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Menyensor teks yang diberikan
|
|
28
|
+
* @param text Teks yang akan disensor
|
|
29
|
+
* @returns Hasil filter
|
|
30
|
+
*/
|
|
31
|
+
filter(text: string): FilterResult {
|
|
32
|
+
return filter(text, this.options);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Memeriksa apakah teks mengandung kata kotor
|
|
37
|
+
* @param text Teks yang akan diperiksa
|
|
38
|
+
* @returns Boolean apakah teks mengandung kata kotor
|
|
39
|
+
*/
|
|
40
|
+
isProfane(text: string): boolean {
|
|
41
|
+
return isProfane(text, this.options);
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Menganalisis teks untuk kata kotor
|
|
46
|
+
* @param text Teks yang akan dianalisis
|
|
47
|
+
* @returns Hasil analisis
|
|
48
|
+
*/
|
|
49
|
+
analyze(text: string): AnalysisResult {
|
|
50
|
+
return analyze(text, this.options);
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Menganalisis batch teks untuk kata kotor
|
|
55
|
+
* @param texts Array dari teks yang akan dianalisis
|
|
56
|
+
* @returns Hasil analisis batch
|
|
57
|
+
*/
|
|
58
|
+
batchAnalyze(texts: string[]) {
|
|
59
|
+
return batchAnalyze(texts, this.options);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Menganalisis teks per-kalimat
|
|
64
|
+
* @param text Teks yang akan dianalisis
|
|
65
|
+
* @returns Array hasil analisis per-kalimat
|
|
66
|
+
*/
|
|
67
|
+
analyzeBySentence(text: string) {
|
|
68
|
+
return analyzeBySentence(text, this.options);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Menganalisis teks dengan konteks di sekitar kata kotor
|
|
73
|
+
* @param text Teks yang akan dianalisis
|
|
74
|
+
* @param contextWindowSize Jumlah kata konteks
|
|
75
|
+
* @returns Konteks di dekat kata kotor
|
|
76
|
+
*/
|
|
77
|
+
analyzeWithContext(text: string, contextWindowSize: number = 5) {
|
|
78
|
+
return analyzeWithContext(text, contextWindowSize, this.options);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Mengubah opsi filter
|
|
83
|
+
* @param options Opsi baru untuk filter
|
|
84
|
+
*/
|
|
85
|
+
setOptions(options: Partial<FilterOptions>) {
|
|
86
|
+
this.options = {
|
|
87
|
+
...this.options,
|
|
88
|
+
...options,
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Menetapkan daftar kata kustom
|
|
94
|
+
* @param wordList Daftar kata untuk digunakan
|
|
95
|
+
*/
|
|
96
|
+
setWordList(wordList: string[]) {
|
|
97
|
+
this.options.wordList = wordList;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Menambahkan kata ke whitelist
|
|
102
|
+
* @param word Kata yang akan diabaikan
|
|
103
|
+
*/
|
|
104
|
+
addToWhitelist(word: string) {
|
|
105
|
+
this.options.whitelist = [
|
|
106
|
+
...(this.options.whitelist || []),
|
|
107
|
+
word.toLowerCase(),
|
|
108
|
+
];
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* Menghapus kata dari whitelist
|
|
113
|
+
* @param word Kata yang akan dihapus dari whitelist
|
|
114
|
+
*/
|
|
115
|
+
removeFromWhitelist(word: string) {
|
|
116
|
+
if (!this.options.whitelist) return;
|
|
117
|
+
|
|
118
|
+
this.options.whitelist = this.options.whitelist.filter(
|
|
119
|
+
(w) => w.toLowerCase() !== word.toLowerCase(),
|
|
120
|
+
);
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
export default IDProfanityFilter;
|
|
125
|
+
|
|
126
|
+
export const idFilter = {
|
|
127
|
+
filter: (text: string, options?: FilterOptions) => filter(text, options),
|
|
128
|
+
isProfane: (text: string, options?: FilterOptions) =>
|
|
129
|
+
isProfane(text, options),
|
|
130
|
+
analyze: (text: string, options?: FilterOptions) => analyze(text, options),
|
|
131
|
+
batchAnalyze: (texts: string[], options?: FilterOptions) =>
|
|
132
|
+
batchAnalyze(texts, options),
|
|
133
|
+
};
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
export type ProfanityCategory =
|
|
2
|
+
| 'sexual' // Kata-kata berbau seksual
|
|
3
|
+
| 'insult' // Kata-kata penghinaan
|
|
4
|
+
| 'profanity' // Umpatan umum
|
|
5
|
+
| 'slur' // Perkataan merendahkan berdasarkan identitas
|
|
6
|
+
| 'drugs' // Terkait narkoba
|
|
7
|
+
| 'disgusting' // Kata-kata menjijikkan
|
|
8
|
+
| 'blasphemy'; // Penistaan agama
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
export type Region =
|
|
12
|
+
| 'general' // Umum di Indonesia
|
|
13
|
+
| 'jawa'
|
|
14
|
+
| 'sunda'
|
|
15
|
+
| 'betawi'
|
|
16
|
+
| 'batak'
|
|
17
|
+
| 'minang'
|
|
18
|
+
| 'bali'
|
|
19
|
+
| 'madura'
|
|
20
|
+
| 'bugis'
|
|
21
|
+
| 'aceh'
|
|
22
|
+
| 'ambon'
|
|
23
|
+
| 'papua'
|
|
24
|
+
| 'manado'
|
|
25
|
+
| 'banjar'
|
|
26
|
+
| 'palembang'
|
|
27
|
+
| 'lampung'
|
|
28
|
+
| 'ntt'
|
|
29
|
+
| 'ntb';
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Definisi kata kotor
|
|
33
|
+
*/
|
|
34
|
+
export interface ProfanityWord {
|
|
35
|
+
word: string; // Kata kotor
|
|
36
|
+
category: ProfanityCategory; // Kategori kata
|
|
37
|
+
region: Region; // Daerah asal kata
|
|
38
|
+
severity: number; // Tingkat keparahan (0-1)
|
|
39
|
+
aliases?: string[]; // Variasi kata atau alias
|
|
40
|
+
description?: string; // Deskripsi atau makna kata
|
|
41
|
+
context?: string; // Konteks penggunaan kata
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Opsi untuk konfigurasi filter
|
|
46
|
+
*/
|
|
47
|
+
export interface FilterOptions {
|
|
48
|
+
/** List kata yang ingin difilter */
|
|
49
|
+
wordList?: string[];
|
|
50
|
+
|
|
51
|
+
/** Karakter pengganti untuk kata yang disensor */
|
|
52
|
+
replaceWith?: string;
|
|
53
|
+
|
|
54
|
+
/** Apakah harus menyensor seluruh kata atau hanya sebagian */
|
|
55
|
+
fullWordCensor?: boolean;
|
|
56
|
+
|
|
57
|
+
/** Apakah harus mendeteksi kata dengan variasi penulisan */
|
|
58
|
+
detectLeetSpeak?: boolean;
|
|
59
|
+
|
|
60
|
+
/** Mengabaikan kata-kata dalam konteks tertentu (misalnya nama) */
|
|
61
|
+
whitelist?: string[];
|
|
62
|
+
|
|
63
|
+
/** Apakah harus memeriksa substring (lebih ketat) */
|
|
64
|
+
checkSubstring?: boolean;
|
|
65
|
+
|
|
66
|
+
/** Kategori kata yang ingin difilter */
|
|
67
|
+
categories?: ProfanityCategory[];
|
|
68
|
+
|
|
69
|
+
/** Daerah yang ingin difilter katanya */
|
|
70
|
+
regions?: Region[];
|
|
71
|
+
|
|
72
|
+
/** Tingkat keparahan minimum yang akan difilter (0-1) */
|
|
73
|
+
severityThreshold?: number;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Hasil analisis dari teks
|
|
78
|
+
*/
|
|
79
|
+
export interface AnalysisResult {
|
|
80
|
+
/** Apakah teks mengandung kata kotor */
|
|
81
|
+
hasProfanity: boolean;
|
|
82
|
+
|
|
83
|
+
/** Daftar kata kotor yang ditemukan */
|
|
84
|
+
matches: string[];
|
|
85
|
+
|
|
86
|
+
/** Daftar kata kotor lengkap dengan metadata */
|
|
87
|
+
matchDetails: ProfanityWord[];
|
|
88
|
+
|
|
89
|
+
/** Daftar kategori kata kotor yang ditemukan */
|
|
90
|
+
categories: ProfanityCategory[];
|
|
91
|
+
|
|
92
|
+
/** Daftar daerah asal kata kotor yang ditemukan */
|
|
93
|
+
regions: Region[];
|
|
94
|
+
|
|
95
|
+
/** Tingkat keparahan (0-1) berdasarkan jumlah dan jenis kata */
|
|
96
|
+
severityScore: number;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Hasil filter dari teks
|
|
101
|
+
*/
|
|
102
|
+
export interface FilterResult {
|
|
103
|
+
/** Teks hasil filter */
|
|
104
|
+
filtered: string;
|
|
105
|
+
|
|
106
|
+
/** Jumlah kata yang disensor */
|
|
107
|
+
censored: number;
|
|
108
|
+
|
|
109
|
+
/** Kata asli yang disensor dan penggantinya */
|
|
110
|
+
replacements: Array<{
|
|
111
|
+
original: string;
|
|
112
|
+
censored: string;
|
|
113
|
+
metadata?: ProfanityWord;
|
|
114
|
+
}>;
|
|
115
|
+
}
|
|
File without changes
|
|
File without changes
|