@sideid/id-profanity-filter 1.0.0 → 1.9.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,203 @@
1
+ import { escapeRegExp } from "./stringUtils";
2
+ import { RegexOptions } from "../types";
3
+
4
+ /**
5
+ * Membuat pola regex untuk mencocokkan kata
6
+ *
7
+ * @param word Kata yang akan dibuat pola regex-nya
8
+ * @param options Opsi untuk pembuatan regex
9
+ * @returns Objek RegExp
10
+ */
11
+ export function createWordRegex(
12
+ word: string,
13
+ options: RegexOptions = {},
14
+ ): RegExp {
15
+ const {
16
+ wholeWord = true,
17
+ caseSensitive = false,
18
+ leetSpeak = true,
19
+ detectSplit = false,
20
+ indonesianVariation = false,
21
+ } = options;
22
+
23
+ // Escape karakter khusus regex
24
+ let pattern = escapeRegExp(word);
25
+
26
+ // Tambahkan variasi leet speak jika diminta
27
+ if (leetSpeak) {
28
+ pattern = addLeetSpeakVariations(pattern);
29
+ }
30
+
31
+ // Tambahkan variasi ejaan Bahasa Indonesia jika diminta
32
+ if (indonesianVariation) {
33
+ pattern = addIndonesianVariations(pattern);
34
+ }
35
+
36
+ // Tambahkan kemungkinan split jika diminta
37
+ if (detectSplit) {
38
+ pattern = addSplitVariations(pattern);
39
+ }
40
+
41
+ // Tambahkan boundary untuk whole word jika diminta
42
+ if (wholeWord) {
43
+ pattern = `\\b${pattern}\\b`;
44
+ }
45
+
46
+ // Buat regex dengan flag case-insensitive jika diminta
47
+ return new RegExp(pattern, caseSensitive ? "g" : "gi");
48
+ }
49
+
50
+ /**
51
+ * Menambahkan variasi leet speak ke pola regex
52
+ *
53
+ * Contoh:
54
+ * - 'a' bisa jadi '4', '@'
55
+ * - 'i' bisa jadi '1', '!'
56
+ *
57
+ * @param pattern Pola regex asli
58
+ * @returns Pola regex dengan variasi leet speak
59
+ */
60
+ export function addLeetSpeakVariations(pattern: string): string {
61
+ const leetMap: Record<string, string[]> = {
62
+ a: ["a", "4", "@"],
63
+ b: ["b", "8", "6"],
64
+ c: ["c", "(", "{", "<"],
65
+ e: ["e", "3"],
66
+ g: ["g", "6", "9"],
67
+ i: ["i", "1", "!", "|"],
68
+ l: ["l", "1", "|"],
69
+ o: ["o", "0"],
70
+ s: ["s", "5", "$"],
71
+ t: ["t", "7", "+"],
72
+ z: ["z", "2"],
73
+ };
74
+
75
+ // Ganti tiap karakter dengan variasinya dalam grup character class
76
+ return pattern
77
+ .split("")
78
+ .map((char) => {
79
+ const lowerChar = char.toLowerCase();
80
+ const variations = leetMap[lowerChar];
81
+
82
+ if (variations && variations.length > 1) {
83
+ return `[${variations.join("")}]`;
84
+ }
85
+
86
+ return char;
87
+ })
88
+ .join("");
89
+ }
90
+
91
+ /**
92
+ * Menambahkan kemungkinan split/pemisahan antar karakter
93
+ *
94
+ * @param pattern Pola regex asli
95
+ * @returns Pola regex dengan kemungkinan split
96
+ */
97
+ export function addSplitVariations(pattern: string): string {
98
+ // Tambahkan kemungkinan spasi atau karakter penghubung di antara setiap huruf
99
+ return pattern.split("").join("[\\s\\-._*+]?");
100
+ }
101
+
102
+ /**
103
+ * Membuat regex untuk mencari kata dengan variasi spasi dan karakter penghubung
104
+ * Berguna untuk mendeteksi upaya menghindari filter dengan menambahkan spasi atau karakter lain
105
+ *
106
+ * @param word Kata yang akan dibuat pola regexnya
107
+ * @returns Objek RegExp
108
+ */
109
+ export function createEvasionRegex(word: string): RegExp {
110
+ // Tambahkan kemungkinan spasi atau karakter penghubung di antara setiap huruf
111
+ const pattern = addSplitVariations(escapeRegExp(word));
112
+
113
+ return new RegExp(pattern, "gi");
114
+ }
115
+
116
+ /**
117
+ * Menambahkan variasi ejaan Bahasa Indonesia
118
+ *
119
+ * @param pattern Pola regex asli
120
+ * @returns Pola regex dengan variasi ejaan Bahasa Indonesia
121
+ */
122
+ export function addIndonesianVariations(pattern: string): string {
123
+ // Variasi ejaan dalam Bahasa Indonesia
124
+ const variationMap: Record<string, string[]> = {
125
+ c: ["c", "k"], // contoh: becok/bekok
126
+ k: ["k", "c", "q"], // contoh: kacau/qacau
127
+ j: ["j", "dj"], // contoh: jualan/djualan (ejaan lama)
128
+ y: ["y", "j"], // contoh: ya/ja
129
+ u: ["u", "oe"], // contoh: untuk/oentoek (ejaan lama)
130
+ f: ["f", "p", "v"], // contoh: kafir/kapir
131
+ z: ["z", "j", "s"], // contoh: zaman/jaman
132
+ x: ["x", "ks"], // contoh: taxi/taksi
133
+ };
134
+
135
+ // Ganti tiap karakter dengan variasinya
136
+ return pattern
137
+ .split("")
138
+ .map((char) => {
139
+ const lowerChar = char.toLowerCase();
140
+ const variations = variationMap[lowerChar];
141
+
142
+ if (variations && variations.length > 1) {
143
+ return `[${variations.join("")}]`;
144
+ }
145
+
146
+ return char;
147
+ })
148
+ .join("");
149
+ }
150
+
151
+ /**
152
+ * Membuat regex untuk mencocokkan kata dengan mempertimbangkan variasi ejaan Bahasa Indonesia
153
+ *
154
+ * @param word Kata yang akan dibuat pola regexnya
155
+ * @returns Objek RegExp
156
+ */
157
+ export function createIndonesianVariationRegex(word: string): RegExp {
158
+ const pattern = addIndonesianVariations(escapeRegExp(word));
159
+ return new RegExp(`\\b${pattern}\\b`, "gi");
160
+ }
161
+
162
+ /**
163
+ * Membuat regex untuk mencocokkan kata dengan konteks
164
+ *
165
+ * @param word Kata yang akan dibuat pola regexnya
166
+ * @param contextSize Jumlah kata konteks sebelum dan sesudah
167
+ * @returns Objek RegExp
168
+ */
169
+ export function createContextRegex(
170
+ word: string,
171
+ contextSize: number = 3,
172
+ ): RegExp {
173
+ const wordPattern = escapeRegExp(word);
174
+
175
+ // Membuat pola yang menangkap beberapa kata sebelum dan setelah kata target
176
+ const pattern = `((?:\\S+\\s+){0,${contextSize}})(\\b${wordPattern}\\b)((?:\\s+\\S+){0,${contextSize}})`;
177
+
178
+ return new RegExp(pattern, "gi");
179
+ }
180
+
181
+ /**
182
+ * Membuat regex untuk mencocokkan variasi penulisan kata
183
+ *
184
+ * @param word Kata dasar
185
+ * @returns Objek RegExp untuk mencocokkan berbagai bentuk kata
186
+ */
187
+ export function createWordFormRegex(word: string): RegExp {
188
+ // Implementasi sederhana untuk mencocokkan berbagai imbuhan
189
+ // Ini bisa dikembangkan lebih lanjut untuk mencocokkan bentukan kata yang lebih kompleks
190
+ const prefixes = ["", "me", "pe", "ber", "di", "ter", "se"];
191
+ const suffixes = ["", "kan", "an", "i", "nya"];
192
+
193
+ const patterns = [];
194
+
195
+ // Kombinasikan prefix dan suffix
196
+ for (const prefix of prefixes) {
197
+ for (const suffix of suffixes) {
198
+ patterns.push(`\\b${prefix}${escapeRegExp(word)}${suffix}\\b`);
199
+ }
200
+ }
201
+
202
+ return new RegExp(patterns.join("|"), "gi");
203
+ }
@@ -0,0 +1,195 @@
1
+ /**
2
+ * Menghitung jarak Levenshtein antara dua string
3
+ * (Jumlah operasi insert, delete, atau replace untuk mengubah string1 menjadi string2)
4
+ *
5
+ * @param str1 String pertama
6
+ * @param str2 String kedua
7
+ * @returns Jarak Levenshtein
8
+ */
9
+ export function levenshteinDistance(str1: string, str2: string): number {
10
+ const s1 = str1.toLowerCase();
11
+ const s2 = str2.toLowerCase();
12
+
13
+ const len1 = s1.length;
14
+ const len2 = s2.length;
15
+
16
+ // Inisialisasi matrix
17
+ const matrix: number[][] = [];
18
+
19
+ // Inisialisasi baris pertama
20
+ for (let i = 0; i <= len2; i++) {
21
+ matrix[0] = matrix[0] || [];
22
+ matrix[0][i] = i;
23
+ }
24
+
25
+ // Inisialisasi kolom pertama
26
+ for (let i = 0; i <= len1; i++) {
27
+ matrix[i] = matrix[i] || [];
28
+ matrix[i][0] = i;
29
+ }
30
+
31
+ // Isi matrix
32
+ for (let i = 1; i <= len1; i++) {
33
+ for (let j = 1; j <= len2; j++) {
34
+ const cost = s1[i - 1] === s2[j - 1] ? 0 : 1;
35
+
36
+ matrix[i][j] = Math.min(
37
+ matrix[i - 1][j] + 1, // deletion
38
+ matrix[i][j - 1] + 1, // insertion
39
+ matrix[i - 1][j - 1] + cost, // substitution
40
+ );
41
+ }
42
+ }
43
+
44
+ return matrix[len1][len2];
45
+ }
46
+
47
+ /**
48
+ * Menghitung tingkat kesamaan antara dua string
49
+ *
50
+ * @param str1 String pertama
51
+ * @param str2 String kedua
52
+ * @returns Nilai kesamaan (0-1, di mana 1 berarti identik)
53
+ */
54
+ export function stringSimilarity(str1: string, str2: string): number {
55
+ if (!str1.length && !str2.length) return 1;
56
+ if (!str1.length || !str2.length) return 0;
57
+
58
+ const distance = levenshteinDistance(str1, str2);
59
+ const maxLength = Math.max(str1.length, str2.length);
60
+
61
+ return 1 - distance / maxLength;
62
+ }
63
+
64
+ /**
65
+ * Mencari string yang paling mirip dari array
66
+ *
67
+ * @param target String target
68
+ * @param candidates Array string kandidat
69
+ * @param threshold Minimum kesamaan yang diterima (0-1)
70
+ * @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
71
+ */
72
+ export function findMostSimilar(
73
+ target: string,
74
+ candidates: string[],
75
+ threshold: number = 0.7,
76
+ ): string | null {
77
+ if (!candidates.length) return null;
78
+
79
+ let maxSimilarity = 0;
80
+ let mostSimilar: string | null = null;
81
+
82
+ for (const candidate of candidates) {
83
+ const similarity = stringSimilarity(target, candidate);
84
+
85
+ if (similarity > maxSimilarity && similarity >= threshold) {
86
+ maxSimilarity = similarity;
87
+ mostSimilar = candidate;
88
+ }
89
+ }
90
+
91
+ return mostSimilar;
92
+ }
93
+
94
+ /**
95
+ * Cek apakah string mungkin merupakan variasi dari kata kotor
96
+ * menggunakan kesamaan string
97
+ *
98
+ * @param input String yang akan diperiksa
99
+ * @param profanityWords Daftar kata kotor
100
+ * @param threshold Batas minimum kesamaan (default: 0.75)
101
+ * @returns Array [Boolean (apakah variasi), String original (jika ditemukan)]
102
+ */
103
+ export function isPossibleProfanityVariation(
104
+ input: string,
105
+ profanityWords: string[],
106
+ threshold: number = 0.75,
107
+ ): [boolean, string | null] {
108
+ if (!input || !profanityWords.length) return [false, null];
109
+
110
+ for (const word of profanityWords) {
111
+ const similarity = stringSimilarity(input, word);
112
+
113
+ if (similarity >= threshold) {
114
+ return [true, word];
115
+ }
116
+ }
117
+
118
+ return [false, null];
119
+ }
120
+
121
+ /**
122
+ * Mengelompokkan kata berdasarkan kesamaan
123
+ *
124
+ * @param words Daftar kata
125
+ * @param threshold Batas minimum kesamaan (default: 0.8)
126
+ * @returns Array kluster kata yang mirip
127
+ */
128
+ export function clusterSimilarWords(
129
+ words: string[],
130
+ threshold: number = 0.8,
131
+ ): string[][] {
132
+ const clusters: string[][] = [];
133
+ const processed: Set<string> = new Set();
134
+
135
+ for (const word of words) {
136
+ if (processed.has(word)) continue;
137
+
138
+ const cluster: string[] = [word];
139
+ processed.add(word);
140
+
141
+ for (const otherWord of words) {
142
+ if (word === otherWord || processed.has(otherWord)) continue;
143
+
144
+ const similarity = stringSimilarity(word, otherWord);
145
+ if (similarity >= threshold) {
146
+ cluster.push(otherWord);
147
+ processed.add(otherWord);
148
+ }
149
+ }
150
+
151
+ clusters.push(cluster);
152
+ }
153
+
154
+ return clusters;
155
+ }
156
+
157
+ /**
158
+ * Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
159
+ *
160
+ * @param text Teks yang akan diperiksa
161
+ * @param profanityWords Daftar kata kotor
162
+ * @param threshold Batas minimum kesamaan (default: 0.8)
163
+ * @returns Array kata yang mungkin merupakan kata kotor
164
+ */
165
+ export function findPossibleProfanityBySimiliarity(
166
+ text: string,
167
+ profanityWords: string[],
168
+ threshold: number = 0.8,
169
+ ): Array<{ word: string; original: string; similarity: number }> {
170
+ const result: Array<{ word: string; original: string; similarity: number }> =
171
+ [];
172
+
173
+ // Pisahkan teks menjadi kata-kata
174
+ const words = text.toLowerCase().split(/\s+/);
175
+
176
+ for (const word of words) {
177
+ // Lewati kata-kata yang terlalu pendek
178
+ if (word.length < 3) continue;
179
+
180
+ for (const profanity of profanityWords) {
181
+ const similarity = stringSimilarity(word, profanity);
182
+
183
+ if (similarity >= threshold) {
184
+ result.push({
185
+ word,
186
+ original: profanity,
187
+ similarity,
188
+ });
189
+ break;
190
+ }
191
+ }
192
+ }
193
+
194
+ return result;
195
+ }
@@ -0,0 +1,213 @@
1
+ /**
2
+ * Menyensor kata dengan karakter pengganti
3
+ *
4
+ * @param word Kata yang akan disensor
5
+ * @param replaceChar Karakter pengganti (default: '*')
6
+ * @param keepFirstAndLast Apakah harus menyimpan huruf pertama dan terakhir
7
+ * @returns Kata yang telah disensor
8
+ */
9
+ export function censorWord(
10
+ word: string,
11
+ replaceChar: string = "*",
12
+ keepFirstAndLast: boolean = false,
13
+ ): string {
14
+ if (word.length <= 2) {
15
+ return replaceChar.repeat(word.length);
16
+ }
17
+
18
+ if (keepFirstAndLast) {
19
+ return `${word[0]}${replaceChar.repeat(word.length - 2)}${word[word.length - 1]}`;
20
+ }
21
+
22
+ return replaceChar.repeat(word.length);
23
+ }
24
+
25
+ /**
26
+ * Menormalisasi string untuk keperluan pembandingan
27
+ *
28
+ * @param text Teks yang akan dinormalisasi
29
+ * @returns Teks yang telah dinormalisasi
30
+ */
31
+ export function normalizeText(text: string): string {
32
+ return text
33
+ .toLowerCase()
34
+ .normalize("NFD") // Normalisasi Unicode
35
+ .replace(/[\u0300-\u036f]/g, "") // Hapus diacritic marks
36
+ .replace(/[^\w\s]/g, "") // Hapus karakter non-alphanumeric
37
+ .trim(); // Hapus whitespace di awal dan akhir
38
+ }
39
+
40
+ /**
41
+ * Mendeteksi apakah string berisi sebagian atau keseluruhan kata dalam wordList
42
+ *
43
+ * @param text Teks yang akan diperiksa
44
+ * @param wordList Daftar kata yang dicari
45
+ * @param checkSubstring Apakah harus memeriksa substring
46
+ * @returns Boolean apakah teks mengandung kata-kata dalam wordList
47
+ */
48
+ export function containsAnyWord(
49
+ text: string,
50
+ wordList: string[],
51
+ checkSubstring: boolean = false,
52
+ ): boolean {
53
+ const normalizedText = normalizeText(text);
54
+
55
+ return wordList.some((word) => {
56
+ const normalizedWord = normalizeText(word);
57
+ return checkSubstring
58
+ ? normalizedText.includes(normalizedWord)
59
+ : new RegExp(`\\b${escapeRegExp(normalizedWord)}\\b`, "i").test(
60
+ normalizedText,
61
+ );
62
+ });
63
+ }
64
+
65
+ /**
66
+ * Memerikasa apakah stirng merupakan kode untuk kata kotor
67
+ * (Menangkap kasus seperti disensor dengan titik atau garis: a**ing, b*bi, dll)
68
+ *
69
+ * @param text Teks yang akan diperika
70
+ * @param wordList daftar kata kotor
71
+ * @return Boolean apakah teks mengandung kata kotor
72
+ */
73
+ export function containsEuphemism(text: string, wordList: string[]): boolean {
74
+ return wordList.some((word) => {
75
+ if (word.length <= 2) return false;
76
+
77
+ const firstChar = word[0];
78
+ const lastChar = word[word.length - 1];
79
+ const pattern = new RegExp(
80
+ `\\b${escapeRegExp(firstChar)}[*@#\\-_.!?\\s]{${word.length - 2}}${escapeRegExp(lastChar)}\\b`,
81
+ "i",
82
+ );
83
+
84
+ return pattern.test(text);
85
+ });
86
+ }
87
+
88
+ /**
89
+ * Mendeteksi upaya menghindari filter dengan pemisahan kata
90
+ * (Tangkap kasus seperti: a n j i n g, b-a-b-i, dll)
91
+ *
92
+ * @param text Teks yang akan diperiksa
93
+ * @param wordList Daftar kata kotor
94
+ * @returns Boolean apakah teks mengandung upaya menghindari filter
95
+ */
96
+ export function detectSplitWords(text: string, wordList: string[]): boolean {
97
+ const compressedText = text.replace(/[\s\-_.!?*]/g, "").toLowerCase();
98
+
99
+ return wordList.some((word) => compressedText.includes(normalizeText(word)));
100
+ }
101
+
102
+ export function escapeRegExp(string: string): string {
103
+ return string.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
104
+ }
105
+
106
+ /**
107
+ * Mengganti sebagian kata dengan masking
108
+ * (berguna untuk email, nomor telepon, dll)
109
+ *
110
+ * @param text Teks untuk dimasking
111
+ * @param visibleStart Jumlah karakter yang terlihat di awal
112
+ * @param visibleEnd Jumlah karakter yang terlihat di akhir
113
+ * @param maskChar Karakter masking
114
+ * @returns Teks yang telah dimasking
115
+ */
116
+ export function maskText(
117
+ text: string,
118
+ visibleStart: number = 1,
119
+ visibleEnd: number = 1,
120
+ maskChar: string = "*",
121
+ ): string {
122
+ if (!text) return "";
123
+ if (text.length <= visibleStart + visibleEnd) return text;
124
+
125
+ const start = text.substring(0, visibleStart);
126
+ const middle = maskChar.repeat(text.length - visibleStart - visibleEnd);
127
+ const end = text.substring(text.length - visibleEnd);
128
+
129
+ return start + middle + end;
130
+ }
131
+
132
+ /**
133
+ * menguabh sting menjadi bentuk leet speak
134
+ * (untuk testing filter bypass)
135
+ *
136
+ * @param text Teks yang akan diubah
137
+ * @return Teks yang telah diubah ke leet speak
138
+ */
139
+ export function toLeetSpeak(text: string): string {
140
+ const leetMap: Record<string, string[]> = {
141
+ a: ["4", "@"],
142
+ b: ["8", "6"],
143
+ c: ["<", "(", "{"],
144
+ e: ["3"],
145
+ g: ["9"],
146
+ i: ["1", "!"],
147
+ l: ["1", "|"],
148
+ o: ["0"],
149
+ s: ["5", "$"],
150
+ t: ["7", "+"],
151
+ z: ["2"],
152
+ };
153
+
154
+ return text
155
+ .split("")
156
+ .map((char) => {
157
+ const lowerChar = char.toLowerCase();
158
+ return leetMap[lowerChar] || char;
159
+ })
160
+ .join("");
161
+ }
162
+
163
+ /**
164
+ * memisahkan teks menajdi kelimat
165
+ *
166
+ * @param text Teks yang akan dipisahkan
167
+ * @return Array kalimat yang telah dipisahkan
168
+ */
169
+ export function splitIntoSentences(text: string): string[] {
170
+ // split berdasarkan titik, seru, taya yagn diikuti spasi atau akhir string
171
+ return text
172
+ .split(/(?<=[.!?])\s+|(?<=[.!?])$/)
173
+ .filter((sentence) => sentence.trim().length > 0);
174
+ }
175
+
176
+ /**
177
+ * mengambil kata-kata di sekitar indeks tertentu
178
+ *
179
+ * @param text Teks yang akan diambil
180
+ * @param index Indeks dalam teks
181
+ * @param windowSize jumlah kata di sekitar indeks
182
+ * @return Kata-kata di sekitar indeks
183
+ */
184
+ export function getContextAroundIndex(
185
+ text: string,
186
+ index: number,
187
+ windowSize: number = 5,
188
+ ): string {
189
+ if (!text || index < 0 || index >= text.length) return "";
190
+
191
+ const words = text.split(/\s+/);
192
+
193
+ let currentPosition = 0;
194
+ let targetWordIndex = -1;
195
+
196
+ for (let i = 0; i < words.length; i++) {
197
+ const wordLength = words[i].length;
198
+ if (index >= currentPosition && index < currentPosition + wordLength) {
199
+ targetWordIndex = i;
200
+ break;
201
+ }
202
+ // Tambahkan panjang kata dan spasi
203
+ currentPosition += wordLength + 1;
204
+ }
205
+
206
+ if (targetWordIndex === -1) return "";
207
+
208
+ // Ambil kata-kata di sekitar
209
+ const startIndex = Math.max(0, targetWordIndex - windowSize);
210
+ const endIndex = Math.min(words.length, targetWordIndex + windowSize + 1);
211
+
212
+ return words.slice(startIndex, endIndex).join(" ");
213
+ }
package/tsconfig.json CHANGED
@@ -11,7 +11,7 @@
11
11
  // "disableReferencedProjectLoad": true, /* Reduce the number of projects loaded automatically by TypeScript. */
12
12
 
13
13
  /* Language and Environment */
14
- "target": "es2018" /* Set the JavaScript language version for emitted JavaScript and include compatible library declarations. */,
14
+ "target": "es2020" /* Set the JavaScript language version for emitted JavaScript and include compatible library declarations. */,
15
15
  // "lib": [], /* Specify a set of bundled library declaration files that describe the target runtime environment. */
16
16
  // "jsx": "preserve", /* Specify what JSX code is generated. */
17
17
  // "libReplacement": true, /* Enable lib replacement. */
File without changes
File without changes