@sideid/id-profanity-filter 1.10.5 → 1.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,179 @@
1
+ /**
2
+ * Implementasi algoritma Aho-Corasick untuk pencocokan string
3
+ * Algoritma ini memungkinkan pencarian banyak pola dalam satu kali pembacaan teks
4
+ */
5
+
6
+ interface AhoCorasickNode {
7
+ children: Map<string, AhoCorasickNode>;
8
+ fail: AhoCorasickNode | null;
9
+ output: Set<string>;
10
+ depth: number;
11
+ char?: string;
12
+ }
13
+
14
+ export class AhoCorasick {
15
+ private root: AhoCorasickNode;
16
+ private built: boolean = false;
17
+
18
+ constructor() {
19
+ this.root = {
20
+ children: new Map(),
21
+ fail: null,
22
+ output: new Set(),
23
+ depth: 0,
24
+ };
25
+ }
26
+
27
+ /**
28
+ * Menambahkan pola ke dalam trie
29
+ * @param pattern Pola yang akan ditambahkan
30
+ */
31
+ addPattern(pattern: string): void {
32
+ if (this.built) {
33
+ throw new Error("Cannot add patterns after the automaton is built");
34
+ }
35
+
36
+ let node = this.root;
37
+ const normalizedPattern = pattern.toLowerCase();
38
+
39
+ for (let i = 0; i < normalizedPattern.length; i++) {
40
+ const char = normalizedPattern[i];
41
+
42
+ if (!node.children.has(char)) {
43
+ node.children.set(char, {
44
+ children: new Map(),
45
+ fail: null,
46
+ output: new Set(),
47
+ depth: node.depth + 1,
48
+ char,
49
+ });
50
+ }
51
+
52
+ node = node.children.get(char)!;
53
+ }
54
+
55
+ node.output.add(normalizedPattern);
56
+ }
57
+
58
+ /**
59
+ * Membangun fungsi failure
60
+ */
61
+ build(): void {
62
+ if (this.built) return;
63
+
64
+ const queue: AhoCorasickNode[] = [];
65
+
66
+ // Set fail pointer for depth 1 nodes to root
67
+ for (const child of this.root.children.values()) {
68
+ child.fail = this.root;
69
+ queue.push(child);
70
+ }
71
+
72
+ // BFS to build failure links
73
+ while (queue.length > 0) {
74
+ const current = queue.shift()!;
75
+
76
+ for (const [char, child] of current.children.entries()) {
77
+ queue.push(child);
78
+
79
+ let failNode = current.fail;
80
+
81
+ // Find the longest proper suffix that is also a prefix
82
+ while (failNode !== null && !failNode.children.has(char)) {
83
+ failNode = failNode.fail;
84
+ }
85
+
86
+ if (failNode === null) {
87
+ child.fail = this.root;
88
+ } else {
89
+ child.fail = failNode.children.get(char)!;
90
+
91
+ // Add outputs from the fail state to this node
92
+ for (const output of child.fail.output) {
93
+ child.output.add(output);
94
+ }
95
+ }
96
+ }
97
+ }
98
+
99
+ this.built = true;
100
+ }
101
+
102
+ /**
103
+ * Mencari semua kemunculan pola dalam teks
104
+ * @param text Teks yang akan dicari
105
+ * @returns Map pola yang ditemukan dengan jumlah kemunculannya
106
+ */
107
+ search(text: string): Map<string, number> {
108
+ if (!this.built) {
109
+ this.build();
110
+ }
111
+
112
+ const matches = new Map<string, number>();
113
+ const normalizedText = text.toLowerCase();
114
+ let node = this.root;
115
+
116
+ for (let i = 0; i < normalizedText.length; i++) {
117
+ const char = normalizedText[i];
118
+
119
+ // Follow failure links until we find a matching transition or reach root
120
+ while (node !== this.root && !node.children.has(char)) {
121
+ node = node.fail!;
122
+ }
123
+
124
+ // Try to follow the transition
125
+ if (node.children.has(char)) {
126
+ node = node.children.get(char)!;
127
+ }
128
+
129
+ // Check for any matches at this node
130
+ for (const match of node.output) {
131
+ matches.set(match, (matches.get(match) || 0) + 1);
132
+ }
133
+ }
134
+
135
+ return matches;
136
+ }
137
+
138
+ /**
139
+ * Mencari semua kemunculan pola dalam teks dan mengembalikan hanya pola unik
140
+ * @param text Teks yang akan dicari
141
+ * @returns Set pola yang ditemukan
142
+ */
143
+ searchUnique(text: string): Set<string> {
144
+ const matches = this.search(text);
145
+ return new Set(matches.keys());
146
+ }
147
+
148
+ /**
149
+ * Mengecek apakah teks mengandung setidaknya satu pola
150
+ * @param text Teks yang akan dicari
151
+ * @returns Boolean apakah pola ditemukan
152
+ */
153
+ containsAny(text: string): boolean {
154
+ if (!this.built) {
155
+ this.build();
156
+ }
157
+
158
+ const normalizedText = text.toLowerCase();
159
+ let node = this.root;
160
+
161
+ for (let i = 0; i < normalizedText.length; i++) {
162
+ const char = normalizedText[i];
163
+
164
+ while (node !== this.root && !node.children.has(char)) {
165
+ node = node.fail!;
166
+ }
167
+
168
+ if (node.children.has(char)) {
169
+ node = node.children.get(char)!;
170
+ }
171
+
172
+ if (node.output.size > 0) {
173
+ return true;
174
+ }
175
+ }
176
+
177
+ return false;
178
+ }
179
+ }
@@ -198,12 +198,22 @@ export function clusterSimilarWords(
198
198
 
199
199
  /**
200
200
  * Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
201
+ * dengan optimasi untuk mengurangi kompleksitas
201
202
  *
202
203
  * @param text Teks yang akan diperiksa
203
204
  * @param profanityWords Daftar kata kotor
204
205
  * @param threshold Batas minimum kesamaan (default: 0.8)
205
206
  * @returns Array kata yang mungkin merupakan kata kotor
206
207
  */
208
+ /**
209
+ * Cari kata-kata kotor yang mungkin dari teks berdasarkan kemiripan string
210
+ * dengan optimasi biar prosesnya nggak terlalu berat
211
+ *
212
+ * @param text Teks yang mau dicek
213
+ * @param profanityWords Daftar kata-kata kotor/kasar
214
+ * @param threshold Batas minimal kemiripan (default: 0.8)
215
+ * @returns Array kata yang kemungkinan kata kotor/kasar
216
+ */
207
217
  export function findPossibleProfanityBySimiliarity(
208
218
  text: string,
209
219
  profanityWords: string[],
@@ -212,23 +222,74 @@ export function findPossibleProfanityBySimiliarity(
212
222
  const result: Array<{ word: string; original: string; similarity: number }> =
213
223
  [];
214
224
 
225
+ // Optimasi dengan bikin map kata-kata kotor berdasarkan huruf pertama
226
+ const profanityMap = new Map<string, string[]>();
227
+
228
+ // Kelompokkan kata-kata kotor berdasarkan huruf pertama biar pencariannya lebih cepat
229
+ for (const word of profanityWords) {
230
+ if (word.length < 1) continue;
231
+
232
+ const firstChar = word[0].toLowerCase();
233
+ if (!profanityMap.has(firstChar)) {
234
+ profanityMap.set(firstChar, []);
235
+ }
236
+ profanityMap.get(firstChar)!.push(word);
237
+ }
238
+
215
239
  const words = text.toLowerCase().split(/\s+/);
216
240
 
217
241
  for (const word of words) {
218
242
  if (word.length < 3) continue;
219
243
 
220
- for (const profanity of profanityWords) {
244
+ // Cuma bandingin dengan kata-kata kotor yang huruf pertamanya sama
245
+ // atau yang perbedaan panjangnya masih masuk akal
246
+ const firstChar = word[0];
247
+ const candidateWords = profanityMap.get(firstChar) || [];
248
+
249
+ // Cek juga huruf-huruf yang berdekatan (buat antisipasi typo di huruf pertama)
250
+ // Ini opsional tapi bikin deteksinya lebih bagus
251
+ const charCode = firstChar.charCodeAt(0);
252
+ const prevChar = String.fromCharCode(charCode - 1);
253
+ const nextChar = String.fromCharCode(charCode + 1);
254
+
255
+ const adjacentCandidates = [
256
+ ...(profanityMap.get(prevChar) || []),
257
+ ...(profanityMap.get(nextChar) || []),
258
+ ];
259
+
260
+ // Gabungin kandidat-kandidatnya, prioritasin yang huruf pertamanya sama persis
261
+ const allCandidates = [...candidateWords, ...adjacentCandidates];
262
+
263
+ // Filter kandidat berdasarkan perbedaan panjang sebelum ngitung kemiripannya
264
+ const lengthFilteredCandidates = allCandidates.filter(
265
+ (candidate) => Math.abs(candidate.length - word.length) <= 2,
266
+ );
267
+
268
+ // Cari yang paling cocok
269
+ let bestMatch: {
270
+ word: string;
271
+ original: string;
272
+ similarity: number;
273
+ } | null = null;
274
+
275
+ for (const profanity of lengthFilteredCandidates) {
221
276
  const similarity = stringSimilarity(word, profanity);
222
277
 
223
- if (similarity >= threshold) {
224
- result.push({
278
+ if (
279
+ similarity >= threshold &&
280
+ (!bestMatch || similarity > bestMatch.similarity)
281
+ ) {
282
+ bestMatch = {
225
283
  word,
226
284
  original: profanity,
227
285
  similarity,
228
- });
229
- break;
286
+ };
230
287
  }
231
288
  }
289
+
290
+ if (bestMatch) {
291
+ result.push(bestMatch);
292
+ }
232
293
  }
233
294
 
234
295
  return result;
@@ -261,30 +322,106 @@ export function findProfanityByLevenshteinDistance(
261
322
  distance: number;
262
323
  }> = [];
263
324
 
325
+ // map kata-kata kotor dikelompokkan sesuai panjangnya
326
+ const profanityByLength = new Map<number, string[]>();
327
+
328
+ // Kelompokkin kata-kata kotor berdasarkan panjangnya biar pencarian lebih cepat
329
+ for (const word of profanityWords) {
330
+ const length = word.length;
331
+ if (!profanityByLength.has(length)) {
332
+ profanityByLength.set(length, []);
333
+ }
334
+ profanityByLength.get(length)!.push(word);
335
+ }
336
+
264
337
  const words = text.toLowerCase().split(/\s+/);
265
338
 
266
339
  for (const word of words) {
267
340
  if (word.length < 3) continue;
268
341
 
269
- for (const profanity of profanityWords) {
270
- if (Math.abs(word.length - profanity.length) > maxDistance) continue;
271
-
272
- const distance = levenshteinDistance(word, profanity);
273
- if (distance <= maxDistance) {
274
- const similarity = stringSimilarity(word, profanity);
275
-
276
- if (similarity >= threshold) {
277
- result.push({
278
- word,
279
- original: profanity,
280
- similarity,
281
- distance,
282
- });
283
- break;
342
+ let bestMatch: {
343
+ word: string;
344
+ original: string;
345
+ similarity: number;
346
+ distance: number;
347
+ } | null = null;
348
+
349
+ for (
350
+ let len = Math.max(3, word.length - maxDistance);
351
+ len <= word.length + maxDistance;
352
+ len++
353
+ ) {
354
+ const candidates = profanityByLength.get(len) || [];
355
+
356
+ for (const profanity of candidates) {
357
+ if (!isCharacterCountSimilar(word, profanity, maxDistance)) {
358
+ continue;
359
+ }
360
+
361
+ const distance = levenshteinDistance(word, profanity);
362
+
363
+ if (distance <= maxDistance) {
364
+ const similarity =
365
+ 1 - distance / Math.max(word.length, profanity.length);
366
+
367
+ if (
368
+ similarity >= threshold &&
369
+ (!bestMatch || similarity > bestMatch.similarity)
370
+ ) {
371
+ bestMatch = {
372
+ word,
373
+ original: profanity,
374
+ similarity,
375
+ distance,
376
+ };
377
+
378
+ if (distance === 0 || similarity > 0.95) {
379
+ break;
380
+ }
381
+ }
284
382
  }
285
383
  }
286
384
  }
385
+
386
+ if (bestMatch) {
387
+ result.push(bestMatch);
388
+ }
287
389
  }
288
390
 
289
391
  return result;
290
392
  }
393
+
394
+ /**
395
+ * Helper function to efficiently check if character counts between two strings
396
+ * are similar enough to warrant a full Levenshtein calculation
397
+ */
398
+ function isCharacterCountSimilar(
399
+ str1: string,
400
+ str2: string,
401
+ maxDifference: number,
402
+ ): boolean {
403
+ const charCount1: Record<string, number> = {};
404
+ const charCount2: Record<string, number> = {};
405
+
406
+ for (const char of str1) {
407
+ charCount1[char] = (charCount1[char] || 0) + 1;
408
+ }
409
+
410
+ for (const char of str2) {
411
+ charCount2[char] = (charCount2[char] || 0) + 1;
412
+ }
413
+
414
+ let diffCount = 0;
415
+
416
+ for (const char in charCount1) {
417
+ diffCount += Math.abs((charCount1[char] || 0) - (charCount2[char] || 0));
418
+ }
419
+
420
+ for (const char in charCount2) {
421
+ if (!charCount1[char]) {
422
+ diffCount += charCount2[char];
423
+ }
424
+ }
425
+
426
+ return diffCount <= maxDifference * 2;
427
+ }
package/test.js DELETED
@@ -1,184 +0,0 @@
1
- // full-example.js
2
- const { IDProfanityFilter, idFilter } = require('./dist');
3
-
4
- /**
5
- * Contoh Penggunaan Lengkap ID-Profanity-Filter
6
- * ============================================
7
- * File ini mencakup semua contoh penggunaan utama library.
8
- */
9
-
10
- console.log('============ CONTOH PENGGUNAAN DASAR ============\n');
11
-
12
- const filter = new IDProfanityFilter();
13
-
14
- const teks =
15
- 'Dasar kontol kontol kntl babi asu ngentod kamu, jangan banyak bacot! perek';
16
-
17
- // ===== 1. Cek apakah teks mengandung kata kotor =====
18
- const hasProfanity = filter.isProfane(teks);
19
- console.log('1. Apakah teks mengandung kata kotor?', hasProfanity);
20
-
21
- // ===== 2. Filter kata kotor (mengganti dengan sensor) =====
22
- const hasil = filter.filter(teks);
23
- console.log('\n2. Hasil filter:');
24
- console.log('- Teks asli:', teks);
25
- console.log('- Teks tersensor:', hasil.filtered);
26
- console.log('- Jumlah kata yang disensor:', hasil.censored);
27
- console.log('- Detail penggantian:', hasil.replacements);
28
-
29
- // ===== 3. Analisis konten =====
30
- const analisis = filter.analyze(teks);
31
- console.log('\n3. Hasil analisis:');
32
- console.log('- Mengandung kata kotor:', analisis.hasProfanity);
33
- console.log('- Kata kotor yang ditemukan:', analisis.matches);
34
- console.log('- Kategori kata kotor:', analisis.categories);
35
- console.log('- Daerah asal kata kotor:', analisis.regions);
36
- console.log('- Skor keparahan:', analisis.severityScore.toFixed(2));
37
-
38
- // console.log('\n============ PENGGUNAAN PRESET ============\n');
39
-
40
- // // ===== 4. Menggunakan preset filter =====
41
- // console.log('4. Menggunakan preset filter:');
42
-
43
- // // Preset strict
44
- // filter.usePreset('strict');
45
- // console.log('- Preset strict:', filter.filter(teks).filtered);
46
-
47
- // // Preset childSafe
48
- // filter.usePreset('childSafe');
49
- // console.log('- Preset childSafe:', filter.filter(teks).filtered);
50
-
51
- // // Preset light
52
- // filter.usePreset('light');
53
- // console.log('- Preset light:', filter.filter(teks).filtered);
54
-
55
- console.log('\n============ KUSTOMISASI FILTER ============\n');
56
-
57
- // ===== 5. Kustomisasi filter =====
58
- console.log('5. Kustomisasi filter:');
59
-
60
- filter.setOptions({
61
- replaceWith: '#', // Menggunakan # sebagai karakter pengganti
62
- fullWordCensor: false, // Hanya menyensor sebagian kata
63
- keepFirstAndLast: true, // Menyimpan huruf pertama dan terakhir
64
- });
65
-
66
- console.log('- Filter dengan opsi kustom:', filter.filter(teks).filtered);
67
-
68
- // Mendeteksi variasi penulisan
69
- filter.setOptions({
70
- detectLeetSpeak: true,
71
- indonesianVariation: true,
72
- detectSplit: true,
73
- detectSimilarity: true,
74
- useLevenshtein: true,
75
- similarityThreshold: 0.85,
76
- maxLevenshteinDistance: 2,
77
- });
78
-
79
- const teks2 =
80
- 'Dasar b4b1 kamu, j4nc0k! Ngent0d! k-o-n-t-o-l! k o n t o l k-0nt0l! anjiing kontool kwontol';
81
- console.log('- Teks asli dengan variasi:', teks2);
82
- console.log('- Hasil filter variasi:', filter.filter(teks2).filtered);
83
-
84
- // ===== 6. Menggunakan whitelist =====
85
- console.log('\n6. Menggunakan whitelist:');
86
-
87
- filter.addToWhitelist('anjing');
88
- const tekstBinatang =
89
- 'Anjing itu hewan peliharaan yang setia, tidak seperti bajingan itu';
90
-
91
- console.log('- Teks dengan "anjing" dalam konteks binatang:', tekstBinatang);
92
- console.log(
93
- '- Hasil filter dengan whitelist:',
94
- filter.filter(tekstBinatang).filtered,
95
- );
96
-
97
- // ===== 7. Menggunakan daftar kata kustom =====
98
- console.log('\n7. Menggunakan daftar kata kustom:');
99
-
100
- const customBadWords = ['jelek', 'buruk', 'sampah', 'payah'];
101
-
102
- // Reset filter dan gunakan daftar kustom
103
- filter.setOptions({
104
- wordList: customBadWords,
105
- replaceWith: '*',
106
- });
107
-
108
- const teksKustom = 'Film ini jelek dan payah sekali!';
109
- console.log('- Teks asli:', teksKustom);
110
- console.log('- Hasil filter kustom:', filter.filter(teksKustom).filtered);
111
-
112
- console.log('\n============ ANALISIS LANJUTAN ============\n');
113
-
114
- // ===== 8. Analisis per kalimat =====
115
- console.log('8. Analisis per kalimat:');
116
-
117
- filter.setOptions({}); // Reset ke default
118
- const kalimat =
119
- 'Saya sangat suka filmnya. Tapi pemainnya seperti anjing, aktingnya buruk.';
120
- console.log('- Kalimat:', kalimat);
121
-
122
- const kalimatAnalisis = filter.analyzeBySentence(kalimat);
123
- console.log('- Hasil analisis per kalimat:');
124
- kalimatAnalisis.forEach((hasil, index) => {
125
- console.log(` Kalimat ${index + 1}: "${hasil.sentence}"`);
126
- console.log(` Mengandung kata kotor: ${hasil.hasProfanity}`);
127
- if (hasil.hasProfanity) {
128
- console.log(` Kata kotor: ${hasil.matches.join(', ')}`);
129
- }
130
- });
131
-
132
- // ===== 9. Analisis batch untuk komentar =====
133
- console.log('\n9. Analisis batch:');
134
-
135
- const komentar = [
136
- 'Film ini sangat bagus, ceritanya menarik sekali!',
137
- 'Dasar goblok, sialan kamu!',
138
- 'Anjing emang filmnya, sampah banget.',
139
- ];
140
-
141
- console.log('- Batch teks:');
142
- komentar.forEach((k, i) => console.log(` ${i + 1}. "${k}"`));
143
-
144
- const hasilBatch = filter.batchAnalyze(komentar);
145
- console.log('- Hasil analisis batch:');
146
- console.log(` - Total teks: ${hasilBatch.totalTexts}`);
147
- console.log(` - Teks mengandung kata kotor: ${hasilBatch.profaneTexts}`);
148
- console.log(` - Teks bersih: ${hasilBatch.cleanTexts}`);
149
- console.log(` - Kategori teratas: ${hasilBatch.topCategories.join(', ')}`);
150
- console.log(` - Daerah teratas: ${hasilBatch.topRegions.join(', ')}`);
151
- console.log(' - Kata kotor terbanyak:');
152
- hasilBatch.mostFrequentWords.forEach((word) => {
153
- console.log(` * "${word.word}": ${word.count} kali`);
154
- });
155
-
156
- // ===== 10. Filter berdasarkan kategori dan daerah =====
157
- console.log('\n10. Filter berdasarkan kategori dan daerah:');
158
-
159
- // Filter untuk kata dari daerah Jawa saja
160
- const filterJawa = new IDProfanityFilter({
161
- regions: ['jawa'], // Hanya filter kata dari Jawa
162
- });
163
-
164
- const contohTeksJawa = 'Dasar jancok, asu, bangsat, bacot!';
165
- console.log('- Teks asli:', contohTeksJawa);
166
- console.log(
167
- '- Filter khusus kata Jawa:',
168
- filterJawa.filter(contohTeksJawa).filtered,
169
- );
170
-
171
- // Filter untuk kategori sexual saja
172
- const filterSexual = new IDProfanityFilter({
173
- categories: ['sexual'], // Hanya filter kata kategori seksual
174
- });
175
-
176
- const contohTeksSexual =
177
- 'Kata kotor seperti anjing dan goblok tidak disensor, tapi bokep disensor.';
178
- console.log('- Teks asli:', contohTeksSexual);
179
- console.log(
180
- '- Filter khusus kategori sexual:',
181
- filterSexual.filter(contohTeksSexual).filtered,
182
- );
183
-
184
- console.log('\n============ SELESAI ============');