@sideid/id-profanity-filter 1.1.0 → 1.9.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/config/options.ts +178 -0
- package/src/constants/categories/index.ts +31 -31
- package/src/constants/categories/insult.ts +114 -114
- package/src/constants/categories/sexual.ts +99 -100
- package/src/constants/regions/batak.ts +17 -17
- package/src/constants/regions/betawi.ts +39 -39
- package/src/constants/regions/general.ts +111 -111
- package/src/constants/regions/index.ts +62 -62
- package/src/constants/regions/jawa.ts +102 -102
- package/src/constants/regions/sunda.ts +35 -35
- package/src/constants/wordList.ts +125 -125
- package/src/core/analyzer.ts +225 -204
- package/src/core/filter.ts +129 -129
- package/src/core/matcher.ts +259 -267
- package/src/index.ts +188 -133
- package/src/types/index.ts +88 -115
- package/src/utils/regexUtils.ts +203 -0
- package/src/utils/similarityUtils.ts +195 -0
- package/src/utils/stringUtils.ts +213 -0
package/src/utils/regexUtils.ts
CHANGED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
import { escapeRegExp } from "./stringUtils";
|
|
2
|
+
import { RegexOptions } from "../types";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Membuat pola regex untuk mencocokkan kata
|
|
6
|
+
*
|
|
7
|
+
* @param word Kata yang akan dibuat pola regex-nya
|
|
8
|
+
* @param options Opsi untuk pembuatan regex
|
|
9
|
+
* @returns Objek RegExp
|
|
10
|
+
*/
|
|
11
|
+
export function createWordRegex(
|
|
12
|
+
word: string,
|
|
13
|
+
options: RegexOptions = {},
|
|
14
|
+
): RegExp {
|
|
15
|
+
const {
|
|
16
|
+
wholeWord = true,
|
|
17
|
+
caseSensitive = false,
|
|
18
|
+
leetSpeak = true,
|
|
19
|
+
detectSplit = false,
|
|
20
|
+
indonesianVariation = false,
|
|
21
|
+
} = options;
|
|
22
|
+
|
|
23
|
+
// Escape karakter khusus regex
|
|
24
|
+
let pattern = escapeRegExp(word);
|
|
25
|
+
|
|
26
|
+
// Tambahkan variasi leet speak jika diminta
|
|
27
|
+
if (leetSpeak) {
|
|
28
|
+
pattern = addLeetSpeakVariations(pattern);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// Tambahkan variasi ejaan Bahasa Indonesia jika diminta
|
|
32
|
+
if (indonesianVariation) {
|
|
33
|
+
pattern = addIndonesianVariations(pattern);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
// Tambahkan kemungkinan split jika diminta
|
|
37
|
+
if (detectSplit) {
|
|
38
|
+
pattern = addSplitVariations(pattern);
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
// Tambahkan boundary untuk whole word jika diminta
|
|
42
|
+
if (wholeWord) {
|
|
43
|
+
pattern = `\\b${pattern}\\b`;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// Buat regex dengan flag case-insensitive jika diminta
|
|
47
|
+
return new RegExp(pattern, caseSensitive ? "g" : "gi");
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Menambahkan variasi leet speak ke pola regex
|
|
52
|
+
*
|
|
53
|
+
* Contoh:
|
|
54
|
+
* - 'a' bisa jadi '4', '@'
|
|
55
|
+
* - 'i' bisa jadi '1', '!'
|
|
56
|
+
*
|
|
57
|
+
* @param pattern Pola regex asli
|
|
58
|
+
* @returns Pola regex dengan variasi leet speak
|
|
59
|
+
*/
|
|
60
|
+
export function addLeetSpeakVariations(pattern: string): string {
|
|
61
|
+
const leetMap: Record<string, string[]> = {
|
|
62
|
+
a: ["a", "4", "@"],
|
|
63
|
+
b: ["b", "8", "6"],
|
|
64
|
+
c: ["c", "(", "{", "<"],
|
|
65
|
+
e: ["e", "3"],
|
|
66
|
+
g: ["g", "6", "9"],
|
|
67
|
+
i: ["i", "1", "!", "|"],
|
|
68
|
+
l: ["l", "1", "|"],
|
|
69
|
+
o: ["o", "0"],
|
|
70
|
+
s: ["s", "5", "$"],
|
|
71
|
+
t: ["t", "7", "+"],
|
|
72
|
+
z: ["z", "2"],
|
|
73
|
+
};
|
|
74
|
+
|
|
75
|
+
// Ganti tiap karakter dengan variasinya dalam grup character class
|
|
76
|
+
return pattern
|
|
77
|
+
.split("")
|
|
78
|
+
.map((char) => {
|
|
79
|
+
const lowerChar = char.toLowerCase();
|
|
80
|
+
const variations = leetMap[lowerChar];
|
|
81
|
+
|
|
82
|
+
if (variations && variations.length > 1) {
|
|
83
|
+
return `[${variations.join("")}]`;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
return char;
|
|
87
|
+
})
|
|
88
|
+
.join("");
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Menambahkan kemungkinan split/pemisahan antar karakter
|
|
93
|
+
*
|
|
94
|
+
* @param pattern Pola regex asli
|
|
95
|
+
* @returns Pola regex dengan kemungkinan split
|
|
96
|
+
*/
|
|
97
|
+
export function addSplitVariations(pattern: string): string {
|
|
98
|
+
// Tambahkan kemungkinan spasi atau karakter penghubung di antara setiap huruf
|
|
99
|
+
return pattern.split("").join("[\\s\\-._*+]?");
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Membuat regex untuk mencari kata dengan variasi spasi dan karakter penghubung
|
|
104
|
+
* Berguna untuk mendeteksi upaya menghindari filter dengan menambahkan spasi atau karakter lain
|
|
105
|
+
*
|
|
106
|
+
* @param word Kata yang akan dibuat pola regexnya
|
|
107
|
+
* @returns Objek RegExp
|
|
108
|
+
*/
|
|
109
|
+
export function createEvasionRegex(word: string): RegExp {
|
|
110
|
+
// Tambahkan kemungkinan spasi atau karakter penghubung di antara setiap huruf
|
|
111
|
+
const pattern = addSplitVariations(escapeRegExp(word));
|
|
112
|
+
|
|
113
|
+
return new RegExp(pattern, "gi");
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Menambahkan variasi ejaan Bahasa Indonesia
|
|
118
|
+
*
|
|
119
|
+
* @param pattern Pola regex asli
|
|
120
|
+
* @returns Pola regex dengan variasi ejaan Bahasa Indonesia
|
|
121
|
+
*/
|
|
122
|
+
export function addIndonesianVariations(pattern: string): string {
|
|
123
|
+
// Variasi ejaan dalam Bahasa Indonesia
|
|
124
|
+
const variationMap: Record<string, string[]> = {
|
|
125
|
+
c: ["c", "k"], // contoh: becok/bekok
|
|
126
|
+
k: ["k", "c", "q"], // contoh: kacau/qacau
|
|
127
|
+
j: ["j", "dj"], // contoh: jualan/djualan (ejaan lama)
|
|
128
|
+
y: ["y", "j"], // contoh: ya/ja
|
|
129
|
+
u: ["u", "oe"], // contoh: untuk/oentoek (ejaan lama)
|
|
130
|
+
f: ["f", "p", "v"], // contoh: kafir/kapir
|
|
131
|
+
z: ["z", "j", "s"], // contoh: zaman/jaman
|
|
132
|
+
x: ["x", "ks"], // contoh: taxi/taksi
|
|
133
|
+
};
|
|
134
|
+
|
|
135
|
+
// Ganti tiap karakter dengan variasinya
|
|
136
|
+
return pattern
|
|
137
|
+
.split("")
|
|
138
|
+
.map((char) => {
|
|
139
|
+
const lowerChar = char.toLowerCase();
|
|
140
|
+
const variations = variationMap[lowerChar];
|
|
141
|
+
|
|
142
|
+
if (variations && variations.length > 1) {
|
|
143
|
+
return `[${variations.join("")}]`;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
return char;
|
|
147
|
+
})
|
|
148
|
+
.join("");
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* Membuat regex untuk mencocokkan kata dengan mempertimbangkan variasi ejaan Bahasa Indonesia
|
|
153
|
+
*
|
|
154
|
+
* @param word Kata yang akan dibuat pola regexnya
|
|
155
|
+
* @returns Objek RegExp
|
|
156
|
+
*/
|
|
157
|
+
export function createIndonesianVariationRegex(word: string): RegExp {
|
|
158
|
+
const pattern = addIndonesianVariations(escapeRegExp(word));
|
|
159
|
+
return new RegExp(`\\b${pattern}\\b`, "gi");
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Membuat regex untuk mencocokkan kata dengan konteks
|
|
164
|
+
*
|
|
165
|
+
* @param word Kata yang akan dibuat pola regexnya
|
|
166
|
+
* @param contextSize Jumlah kata konteks sebelum dan sesudah
|
|
167
|
+
* @returns Objek RegExp
|
|
168
|
+
*/
|
|
169
|
+
export function createContextRegex(
|
|
170
|
+
word: string,
|
|
171
|
+
contextSize: number = 3,
|
|
172
|
+
): RegExp {
|
|
173
|
+
const wordPattern = escapeRegExp(word);
|
|
174
|
+
|
|
175
|
+
// Membuat pola yang menangkap beberapa kata sebelum dan setelah kata target
|
|
176
|
+
const pattern = `((?:\\S+\\s+){0,${contextSize}})(\\b${wordPattern}\\b)((?:\\s+\\S+){0,${contextSize}})`;
|
|
177
|
+
|
|
178
|
+
return new RegExp(pattern, "gi");
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* Membuat regex untuk mencocokkan variasi penulisan kata
|
|
183
|
+
*
|
|
184
|
+
* @param word Kata dasar
|
|
185
|
+
* @returns Objek RegExp untuk mencocokkan berbagai bentuk kata
|
|
186
|
+
*/
|
|
187
|
+
export function createWordFormRegex(word: string): RegExp {
|
|
188
|
+
// Implementasi sederhana untuk mencocokkan berbagai imbuhan
|
|
189
|
+
// Ini bisa dikembangkan lebih lanjut untuk mencocokkan bentukan kata yang lebih kompleks
|
|
190
|
+
const prefixes = ["", "me", "pe", "ber", "di", "ter", "se"];
|
|
191
|
+
const suffixes = ["", "kan", "an", "i", "nya"];
|
|
192
|
+
|
|
193
|
+
const patterns = [];
|
|
194
|
+
|
|
195
|
+
// Kombinasikan prefix dan suffix
|
|
196
|
+
for (const prefix of prefixes) {
|
|
197
|
+
for (const suffix of suffixes) {
|
|
198
|
+
patterns.push(`\\b${prefix}${escapeRegExp(word)}${suffix}\\b`);
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
return new RegExp(patterns.join("|"), "gi");
|
|
203
|
+
}
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Menghitung jarak Levenshtein antara dua string
|
|
3
|
+
* (Jumlah operasi insert, delete, atau replace untuk mengubah string1 menjadi string2)
|
|
4
|
+
*
|
|
5
|
+
* @param str1 String pertama
|
|
6
|
+
* @param str2 String kedua
|
|
7
|
+
* @returns Jarak Levenshtein
|
|
8
|
+
*/
|
|
9
|
+
export function levenshteinDistance(str1: string, str2: string): number {
|
|
10
|
+
const s1 = str1.toLowerCase();
|
|
11
|
+
const s2 = str2.toLowerCase();
|
|
12
|
+
|
|
13
|
+
const len1 = s1.length;
|
|
14
|
+
const len2 = s2.length;
|
|
15
|
+
|
|
16
|
+
// Inisialisasi matrix
|
|
17
|
+
const matrix: number[][] = [];
|
|
18
|
+
|
|
19
|
+
// Inisialisasi baris pertama
|
|
20
|
+
for (let i = 0; i <= len2; i++) {
|
|
21
|
+
matrix[0] = matrix[0] || [];
|
|
22
|
+
matrix[0][i] = i;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
// Inisialisasi kolom pertama
|
|
26
|
+
for (let i = 0; i <= len1; i++) {
|
|
27
|
+
matrix[i] = matrix[i] || [];
|
|
28
|
+
matrix[i][0] = i;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// Isi matrix
|
|
32
|
+
for (let i = 1; i <= len1; i++) {
|
|
33
|
+
for (let j = 1; j <= len2; j++) {
|
|
34
|
+
const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
|
|
35
|
+
|
|
36
|
+
matrix[i][j] = Math.min(
|
|
37
|
+
matrix[i - 1][j] + 1, // deletion
|
|
38
|
+
matrix[i][j - 1] + 1, // insertion
|
|
39
|
+
matrix[i - 1][j - 1] + cost, // substitution
|
|
40
|
+
);
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
return matrix[len1][len2];
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Menghitung tingkat kesamaan antara dua string
|
|
49
|
+
*
|
|
50
|
+
* @param str1 String pertama
|
|
51
|
+
* @param str2 String kedua
|
|
52
|
+
* @returns Nilai kesamaan (0-1, di mana 1 berarti identik)
|
|
53
|
+
*/
|
|
54
|
+
export function stringSimilarity(str1: string, str2: string): number {
|
|
55
|
+
if (!str1.length && !str2.length) return 1;
|
|
56
|
+
if (!str1.length || !str2.length) return 0;
|
|
57
|
+
|
|
58
|
+
const distance = levenshteinDistance(str1, str2);
|
|
59
|
+
const maxLength = Math.max(str1.length, str2.length);
|
|
60
|
+
|
|
61
|
+
return 1 - distance / maxLength;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Mencari string yang paling mirip dari array
|
|
66
|
+
*
|
|
67
|
+
* @param target String target
|
|
68
|
+
* @param candidates Array string kandidat
|
|
69
|
+
* @param threshold Minimum kesamaan yang diterima (0-1)
|
|
70
|
+
* @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
|
|
71
|
+
*/
|
|
72
|
+
export function findMostSimilar(
|
|
73
|
+
target: string,
|
|
74
|
+
candidates: string[],
|
|
75
|
+
threshold: number = 0.7,
|
|
76
|
+
): string | null {
|
|
77
|
+
if (!candidates.length) return null;
|
|
78
|
+
|
|
79
|
+
let maxSimilarity = 0;
|
|
80
|
+
let mostSimilar: string | null = null;
|
|
81
|
+
|
|
82
|
+
for (const candidate of candidates) {
|
|
83
|
+
const similarity = stringSimilarity(target, candidate);
|
|
84
|
+
|
|
85
|
+
if (similarity > maxSimilarity && similarity >= threshold) {
|
|
86
|
+
maxSimilarity = similarity;
|
|
87
|
+
mostSimilar = candidate;
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
return mostSimilar;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Cek apakah string mungkin merupakan variasi dari kata kotor
|
|
96
|
+
* menggunakan kesamaan string
|
|
97
|
+
*
|
|
98
|
+
* @param input String yang akan diperiksa
|
|
99
|
+
* @param profanityWords Daftar kata kotor
|
|
100
|
+
* @param threshold Batas minimum kesamaan (default: 0.75)
|
|
101
|
+
* @returns Array [Boolean (apakah variasi), String original (jika ditemukan)]
|
|
102
|
+
*/
|
|
103
|
+
export function isPossibleProfanityVariation(
|
|
104
|
+
input: string,
|
|
105
|
+
profanityWords: string[],
|
|
106
|
+
threshold: number = 0.75,
|
|
107
|
+
): [boolean, string | null] {
|
|
108
|
+
if (!input || !profanityWords.length) return [false, null];
|
|
109
|
+
|
|
110
|
+
for (const word of profanityWords) {
|
|
111
|
+
const similarity = stringSimilarity(input, word);
|
|
112
|
+
|
|
113
|
+
if (similarity >= threshold) {
|
|
114
|
+
return [true, word];
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
return [false, null];
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Mengelompokkan kata berdasarkan kesamaan
|
|
123
|
+
*
|
|
124
|
+
* @param words Daftar kata
|
|
125
|
+
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
126
|
+
* @returns Array kluster kata yang mirip
|
|
127
|
+
*/
|
|
128
|
+
export function clusterSimilarWords(
|
|
129
|
+
words: string[],
|
|
130
|
+
threshold: number = 0.8,
|
|
131
|
+
): string[][] {
|
|
132
|
+
const clusters: string[][] = [];
|
|
133
|
+
const processed: Set<string> = new Set();
|
|
134
|
+
|
|
135
|
+
for (const word of words) {
|
|
136
|
+
if (processed.has(word)) continue;
|
|
137
|
+
|
|
138
|
+
const cluster: string[] = [word];
|
|
139
|
+
processed.add(word);
|
|
140
|
+
|
|
141
|
+
for (const otherWord of words) {
|
|
142
|
+
if (word === otherWord || processed.has(otherWord)) continue;
|
|
143
|
+
|
|
144
|
+
const similarity = stringSimilarity(word, otherWord);
|
|
145
|
+
if (similarity >= threshold) {
|
|
146
|
+
cluster.push(otherWord);
|
|
147
|
+
processed.add(otherWord);
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
clusters.push(cluster);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
return clusters;
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
|
|
159
|
+
*
|
|
160
|
+
* @param text Teks yang akan diperiksa
|
|
161
|
+
* @param profanityWords Daftar kata kotor
|
|
162
|
+
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
163
|
+
* @returns Array kata yang mungkin merupakan kata kotor
|
|
164
|
+
*/
|
|
165
|
+
export function findPossibleProfanityBySimiliarity(
|
|
166
|
+
text: string,
|
|
167
|
+
profanityWords: string[],
|
|
168
|
+
threshold: number = 0.8,
|
|
169
|
+
): Array<{ word: string; original: string; similarity: number }> {
|
|
170
|
+
const result: Array<{ word: string; original: string; similarity: number }> =
|
|
171
|
+
[];
|
|
172
|
+
|
|
173
|
+
// Pisahkan teks menjadi kata-kata
|
|
174
|
+
const words = text.toLowerCase().split(/\s+/);
|
|
175
|
+
|
|
176
|
+
for (const word of words) {
|
|
177
|
+
// Lewati kata-kata yang terlalu pendek
|
|
178
|
+
if (word.length < 3) continue;
|
|
179
|
+
|
|
180
|
+
for (const profanity of profanityWords) {
|
|
181
|
+
const similarity = stringSimilarity(word, profanity);
|
|
182
|
+
|
|
183
|
+
if (similarity >= threshold) {
|
|
184
|
+
result.push({
|
|
185
|
+
word,
|
|
186
|
+
original: profanity,
|
|
187
|
+
similarity,
|
|
188
|
+
});
|
|
189
|
+
break;
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
return result;
|
|
195
|
+
}
|
package/src/utils/stringUtils.ts
CHANGED
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Menyensor kata dengan karakter pengganti
|
|
3
|
+
*
|
|
4
|
+
* @param word Kata yang akan disensor
|
|
5
|
+
* @param replaceChar Karakter pengganti (default: '*')
|
|
6
|
+
* @param keepFirstAndLast Apakah harus menyimpan huruf pertama dan terakhir
|
|
7
|
+
* @returns Kata yang telah disensor
|
|
8
|
+
*/
|
|
9
|
+
export function censorWord(
|
|
10
|
+
word: string,
|
|
11
|
+
replaceChar: string = "*",
|
|
12
|
+
keepFirstAndLast: boolean = false,
|
|
13
|
+
): string {
|
|
14
|
+
if (word.length <= 2) {
|
|
15
|
+
return replaceChar.repeat(word.length);
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
if (keepFirstAndLast) {
|
|
19
|
+
return `${word[0]}${replaceChar.repeat(word.length - 2)}${word[word.length - 1]}`;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
return replaceChar.repeat(word.length);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Menormalisasi string untuk keperluan pembandingan
|
|
27
|
+
*
|
|
28
|
+
* @param text Teks yang akan dinormalisasi
|
|
29
|
+
* @returns Teks yang telah dinormalisasi
|
|
30
|
+
*/
|
|
31
|
+
export function normalizeText(text: string): string {
|
|
32
|
+
return text
|
|
33
|
+
.toLowerCase()
|
|
34
|
+
.normalize("NFD") // Normalisasi Unicode
|
|
35
|
+
.replace(/[\u0300-\u036f]/g, "") // Hapus diacritic marks
|
|
36
|
+
.replace(/[^\w\s]/g, "") // Hapus karakter non-alphanumeric
|
|
37
|
+
.trim(); // Hapus whitespace di awal dan akhir
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Mendeteksi apakah string berisi sebagian atau keseluruhan kata dalam wordList
|
|
42
|
+
*
|
|
43
|
+
* @param text Teks yang akan diperiksa
|
|
44
|
+
* @param wordList Daftar kata yang dicari
|
|
45
|
+
* @param checkSubstring Apakah harus memeriksa substring
|
|
46
|
+
* @returns Boolean apakah teks mengandung kata-kata dalam wordList
|
|
47
|
+
*/
|
|
48
|
+
export function containsAnyWord(
|
|
49
|
+
text: string,
|
|
50
|
+
wordList: string[],
|
|
51
|
+
checkSubstring: boolean = false,
|
|
52
|
+
): boolean {
|
|
53
|
+
const normalizedText = normalizeText(text);
|
|
54
|
+
|
|
55
|
+
return wordList.some((word) => {
|
|
56
|
+
const normalizedWord = normalizeText(word);
|
|
57
|
+
return checkSubstring
|
|
58
|
+
? normalizedText.includes(normalizedWord)
|
|
59
|
+
: new RegExp(`\\b${escapeRegExp(normalizedWord)}\\b`, "i").test(
|
|
60
|
+
normalizedText,
|
|
61
|
+
);
|
|
62
|
+
});
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Memerikasa apakah stirng merupakan kode untuk kata kotor
|
|
67
|
+
* (Menangkap kasus seperti disensor dengan titik atau garis: a**ing, b*bi, dll)
|
|
68
|
+
*
|
|
69
|
+
* @param text Teks yang akan diperika
|
|
70
|
+
* @param wordList daftar kata kotor
|
|
71
|
+
* @return Boolean apakah teks mengandung kata kotor
|
|
72
|
+
*/
|
|
73
|
+
export function containsEuphemism(text: string, wordList: string[]): boolean {
|
|
74
|
+
return wordList.some((word) => {
|
|
75
|
+
if (word.length <= 2) return false;
|
|
76
|
+
|
|
77
|
+
const firstChar = word[0];
|
|
78
|
+
const lastChar = word[word.length - 1];
|
|
79
|
+
const pattern = new RegExp(
|
|
80
|
+
`\\b${escapeRegExp(firstChar)}[*@#\\-_.!?\\s]{${word.length - 2}}${escapeRegExp(lastChar)}\\b`,
|
|
81
|
+
"i",
|
|
82
|
+
);
|
|
83
|
+
|
|
84
|
+
return pattern.test(text);
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Mendeteksi upaya menghindari filter dengan pemisahan kata
|
|
90
|
+
* (Tangkap kasus seperti: a n j i n g, b-a-b-i, dll)
|
|
91
|
+
*
|
|
92
|
+
* @param text Teks yang akan diperiksa
|
|
93
|
+
* @param wordList Daftar kata kotor
|
|
94
|
+
* @returns Boolean apakah teks mengandung upaya menghindari filter
|
|
95
|
+
*/
|
|
96
|
+
export function detectSplitWords(text: string, wordList: string[]): boolean {
|
|
97
|
+
const compressedText = text.replace(/[\s\-_.!?*]/g, "").toLowerCase();
|
|
98
|
+
|
|
99
|
+
return wordList.some((word) => compressedText.includes(normalizeText(word)));
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export function escapeRegExp(string: string): string {
|
|
103
|
+
return string.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Mengganti sebagian kata dengan masking
|
|
108
|
+
* (berguna untuk email, nomor telepon, dll)
|
|
109
|
+
*
|
|
110
|
+
* @param text Teks untuk dimasking
|
|
111
|
+
* @param visibleStart Jumlah karakter yang terlihat di awal
|
|
112
|
+
* @param visibleEnd Jumlah karakter yang terlihat di akhir
|
|
113
|
+
* @param maskChar Karakter masking
|
|
114
|
+
* @returns Teks yang telah dimasking
|
|
115
|
+
*/
|
|
116
|
+
export function maskText(
|
|
117
|
+
text: string,
|
|
118
|
+
visibleStart: number = 1,
|
|
119
|
+
visibleEnd: number = 1,
|
|
120
|
+
maskChar: string = "*",
|
|
121
|
+
): string {
|
|
122
|
+
if (!text) return "";
|
|
123
|
+
if (text.length <= visibleStart + visibleEnd) return text;
|
|
124
|
+
|
|
125
|
+
const start = text.substring(0, visibleStart);
|
|
126
|
+
const middle = maskChar.repeat(text.length - visibleStart - visibleEnd);
|
|
127
|
+
const end = text.substring(text.length - visibleEnd);
|
|
128
|
+
|
|
129
|
+
return start + middle + end;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* menguabh sting menjadi bentuk leet speak
|
|
134
|
+
* (untuk testing filter bypass)
|
|
135
|
+
*
|
|
136
|
+
* @param text Teks yang akan diubah
|
|
137
|
+
* @return Teks yang telah diubah ke leet speak
|
|
138
|
+
*/
|
|
139
|
+
export function toLeetSpeak(text: string): string {
|
|
140
|
+
const leetMap: Record<string, string[]> = {
|
|
141
|
+
a: ["4", "@"],
|
|
142
|
+
b: ["8", "6"],
|
|
143
|
+
c: ["<", "(", "{"],
|
|
144
|
+
e: ["3"],
|
|
145
|
+
g: ["9"],
|
|
146
|
+
i: ["1", "!"],
|
|
147
|
+
l: ["1", "|"],
|
|
148
|
+
o: ["0"],
|
|
149
|
+
s: ["5", "$"],
|
|
150
|
+
t: ["7", "+"],
|
|
151
|
+
z: ["2"],
|
|
152
|
+
};
|
|
153
|
+
|
|
154
|
+
return text
|
|
155
|
+
.split("")
|
|
156
|
+
.map((char) => {
|
|
157
|
+
const lowerChar = char.toLowerCase();
|
|
158
|
+
return leetMap[lowerChar] || char;
|
|
159
|
+
})
|
|
160
|
+
.join("");
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* memisahkan teks menajdi kelimat
|
|
165
|
+
*
|
|
166
|
+
* @param text Teks yang akan dipisahkan
|
|
167
|
+
* @return Array kalimat yang telah dipisahkan
|
|
168
|
+
*/
|
|
169
|
+
export function splitIntoSentences(text: string): string[] {
|
|
170
|
+
// split berdasarkan titik, seru, taya yagn diikuti spasi atau akhir string
|
|
171
|
+
return text
|
|
172
|
+
.split(/(?<=[.!?])\s+|(?<=[.!?])$/)
|
|
173
|
+
.filter((sentence) => sentence.trim().length > 0);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* mengambil kata-kata di sekitar indeks tertentu
|
|
178
|
+
*
|
|
179
|
+
* @param text Teks yang akan diambil
|
|
180
|
+
* @param index Indeks dalam teks
|
|
181
|
+
* @param windowSize jumlah kata di sekitar indeks
|
|
182
|
+
* @return Kata-kata di sekitar indeks
|
|
183
|
+
*/
|
|
184
|
+
export function getContextAroundIndex(
|
|
185
|
+
text: string,
|
|
186
|
+
index: number,
|
|
187
|
+
windowSize: number = 5,
|
|
188
|
+
): string {
|
|
189
|
+
if (!text || index < 0 || index >= text.length) return "";
|
|
190
|
+
|
|
191
|
+
const words = text.split(/\s+/);
|
|
192
|
+
|
|
193
|
+
let currentPosition = 0;
|
|
194
|
+
let targetWordIndex = -1;
|
|
195
|
+
|
|
196
|
+
for (let i = 0; i < words.length; i++) {
|
|
197
|
+
const wordLength = words[i].length;
|
|
198
|
+
if (index >= currentPosition && index < currentPosition + wordLength) {
|
|
199
|
+
targetWordIndex = i;
|
|
200
|
+
break;
|
|
201
|
+
}
|
|
202
|
+
// Tambahkan panjang kata dan spasi
|
|
203
|
+
currentPosition += wordLength + 1;
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
if (targetWordIndex === -1) return "";
|
|
207
|
+
|
|
208
|
+
// Ambil kata-kata di sekitar
|
|
209
|
+
const startIndex = Math.max(0, targetWordIndex - windowSize);
|
|
210
|
+
const endIndex = Math.min(words.length, targetWordIndex + windowSize + 1);
|
|
211
|
+
|
|
212
|
+
return words.slice(startIndex, endIndex).join(" ");
|
|
213
|
+
}
|