@sideid/id-profanity-filter 1.9.5 → 1.10.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,26 +3,21 @@ import {
3
3
  ProfanityCategory,
4
4
  Region,
5
5
  FilterOptions,
6
- } from "../types";
6
+ } from '../types';
7
7
 
8
- import { wordObjects, getWordsByFilter } from "../constants/wordList";
9
- import {
10
- normalizeText,
11
- escapeRegExp,
12
- containsAnyWord,
13
- detectSplitWords,
14
- } from "../utils/stringUtils";
15
- import {
16
- createWordRegex,
17
- addLeetSpeakVariations,
18
- addIndonesianVariations,
19
- addSplitVariations,
20
- } from "../utils/regexUtils";
8
+ import { wordObjects } from '../constants/wordList';
9
+ import { normalizeText } from '../utils/stringUtils';
10
+ import { createWordRegex } from '../utils/regexUtils';
21
11
  import {
22
12
  findPossibleProfanityBySimiliarity,
23
- stringSimilarity,
24
- } from "../utils/similarityUtils";
25
- import { DEFAULT_OPTIONS } from "../config/options";
13
+ findProfanityByLevenshteinDistance,
14
+ } from '../utils/similarityUtils';
15
+ import { DEFAULT_OPTIONS } from '../config/options';
16
+
17
+ interface FindProfanityFunction {
18
+ (text: string, options?: FilterOptions): string[];
19
+ lastActualMatches?: Map<string, string[]>;
20
+ }
26
21
 
27
22
  export function findProfanity(
28
23
  text: string,
@@ -40,38 +35,55 @@ export function findProfanity(
40
35
  detectSimilarity = false,
41
36
  similarityThreshold = 0.8,
42
37
  detectSplit = false,
38
+ useLevenshtein = false,
39
+ maxLevenshteinDistance = 2,
43
40
  } = { ...DEFAULT_OPTIONS, ...options };
44
41
 
45
42
  const normalizedText = normalizeText(text);
46
43
 
47
- let wordsToCheck: string[] = wordList.length > 0 ? wordList : [];
44
+ let baseWordsToCheck: string[] = wordList.length > 0 ? wordList : [];
48
45
 
49
- if (wordsToCheck.length === 0) {
50
- if (categories || regions || severityThreshold > 0) {
51
- wordsToCheck = wordObjects
52
- .filter((word) => {
53
- const matchCategory = categories
54
- ? categories.includes(word.category)
55
- : true;
56
- const matchRegion = regions ? regions.includes(word.region) : true;
57
- const matchSeverity = word.severity >= severityThreshold;
58
- return matchCategory && matchRegion && matchSeverity;
59
- })
60
- .map((word) => word.word);
61
- } else {
62
- wordsToCheck = wordObjects.map((word) => word.word);
63
- }
46
+ if (baseWordsToCheck.length === 0) {
47
+ const filteredWords = wordObjects.filter((word) => {
48
+ const matchCategory = categories
49
+ ? categories.includes(word.category)
50
+ : true;
51
+ const matchRegion = regions ? regions.includes(word.region) : true;
52
+ const matchSeverity = word.severity >= severityThreshold;
53
+ return matchCategory && matchRegion && matchSeverity;
54
+ });
55
+
56
+ baseWordsToCheck = filteredWords.map((word) => word.word);
64
57
  }
65
58
 
66
- wordsToCheck = wordsToCheck.filter(
67
- (word) => !whitelist.includes(word.toLocaleLowerCase()),
68
- );
59
+ const aliasMap = new Map<string, string>();
60
+ wordObjects.forEach((wordObj) => {
61
+ if (wordObj.aliases && wordObj.aliases.length > 0) {
62
+ const matchCategory = categories
63
+ ? categories.includes(wordObj.category)
64
+ : true;
65
+ const matchRegion = regions ? regions.includes(wordObj.region) : true;
66
+ const matchSeverity = wordObj.severity >= severityThreshold;
67
+
68
+ if (matchCategory && matchRegion && matchSeverity) {
69
+ wordObj.aliases.forEach((alias) => {
70
+ aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
71
+ });
72
+ }
73
+ }
74
+ });
75
+
76
+ const wordsToCheck = [
77
+ ...baseWordsToCheck,
78
+ ...Array.from(aliasMap.keys()),
79
+ ].filter((word) => !whitelist.includes(word.toLowerCase()));
69
80
 
70
81
  if (wordsToCheck.length === 0) {
71
82
  return [];
72
83
  }
73
84
 
74
85
  const matches = new Set<string>();
86
+ const actualMatches = new Map<string, string[]>();
75
87
 
76
88
  wordsToCheck.forEach((word) => {
77
89
  const regex = createWordRegex(word, {
@@ -84,7 +96,14 @@ export function findProfanity(
84
96
 
85
97
  let match;
86
98
  while ((match = regex.exec(normalizedText)) !== null) {
87
- matches.add(word.toLowerCase());
99
+ const originalWord =
100
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
101
+ matches.add(originalWord);
102
+
103
+ if (!actualMatches.has(originalWord)) {
104
+ actualMatches.set(originalWord, []);
105
+ }
106
+ actualMatches.get(originalWord)?.push(match[0]);
88
107
  }
89
108
  });
90
109
 
@@ -100,7 +119,14 @@ export function findProfanity(
100
119
 
101
120
  let match;
102
121
  while ((match = leetRegex.exec(text)) !== null) {
103
- matches.add(word.toLowerCase());
122
+ const originalWord =
123
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
124
+ matches.add(originalWord);
125
+
126
+ if (!actualMatches.has(originalWord)) {
127
+ actualMatches.set(originalWord, []);
128
+ }
129
+ actualMatches.get(originalWord)?.push(match[0]);
104
130
  }
105
131
  });
106
132
  }
@@ -117,41 +143,85 @@ export function findProfanity(
117
143
 
118
144
  let match;
119
145
  while ((match = variantRegex.exec(text)) !== null) {
120
- matches.add(word.toLowerCase());
146
+ const originalWord =
147
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
148
+ matches.add(originalWord);
149
+
150
+ if (!actualMatches.has(originalWord)) {
151
+ actualMatches.set(originalWord, []);
152
+ }
153
+ actualMatches.get(originalWord)?.push(match[0]);
121
154
  }
122
155
  });
123
156
  }
124
157
 
125
158
  if (detectSplit) {
126
- if (detectSplitWords(text, wordsToCheck)) {
127
- wordsToCheck.forEach((word) => {
128
- const splitRegex = createWordRegex(word, {
129
- wholeWord: false,
130
- caseSensitive: false,
131
- leetSpeak: false,
132
- detectSplit: true,
133
- indonesianVariation: false,
134
- });
159
+ wordsToCheck.forEach((word) => {
160
+ const splitRegex = createWordRegex(word, {
161
+ wholeWord: false,
162
+ caseSensitive: false,
163
+ leetSpeak: false,
164
+ detectSplit: true,
165
+ indonesianVariation: false,
166
+ });
135
167
 
136
- if (splitRegex.test(text)) {
137
- matches.add(word.toLowerCase());
168
+ let match;
169
+ while ((match = splitRegex.exec(text)) !== null) {
170
+ const originalWord =
171
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
172
+ matches.add(originalWord);
173
+
174
+ if (!actualMatches.has(originalWord)) {
175
+ actualMatches.set(originalWord, []);
138
176
  }
139
- });
140
- }
177
+ actualMatches.get(originalWord)?.push(match[0]);
178
+ }
179
+ });
141
180
  }
142
181
 
143
182
  if (detectSimilarity) {
144
- const possibleProfanity = findPossibleProfanityBySimiliarity(
145
- text,
146
- wordsToCheck,
147
- similarityThreshold,
148
- );
149
-
150
- possibleProfanity.forEach((item) => {
151
- matches.add(item.original.toLowerCase());
152
- });
183
+ if (useLevenshtein) {
184
+ const possibleProfanity = findProfanityByLevenshteinDistance(
185
+ text,
186
+ wordsToCheck,
187
+ similarityThreshold,
188
+ maxLevenshteinDistance,
189
+ );
190
+
191
+ possibleProfanity.forEach((item) => {
192
+ const originalWord =
193
+ aliasMap.get(item.original.toLowerCase()) ||
194
+ item.original.toLowerCase();
195
+ matches.add(originalWord);
196
+
197
+ if (!actualMatches.has(originalWord)) {
198
+ actualMatches.set(originalWord, []);
199
+ }
200
+ actualMatches.get(originalWord)?.push(item.word);
201
+ });
202
+ } else {
203
+ const possibleProfanity = findPossibleProfanityBySimiliarity(
204
+ text,
205
+ wordsToCheck,
206
+ similarityThreshold,
207
+ );
208
+
209
+ possibleProfanity.forEach((item) => {
210
+ matches.add(item.original.toLowerCase());
211
+
212
+ const originalWord =
213
+ aliasMap.get(item.original.toLowerCase()) ||
214
+ item.original.toLowerCase();
215
+ if (!actualMatches.has(originalWord)) {
216
+ actualMatches.set(originalWord, []);
217
+ }
218
+ actualMatches.get(originalWord)?.push(item.word);
219
+ });
220
+ }
153
221
  }
154
222
 
223
+ (findProfanity as FindProfanityFunction).lastActualMatches = actualMatches;
224
+
155
225
  return Array.from(matches);
156
226
  }
157
227
 
package/src/index.ts CHANGED
@@ -1,28 +1,27 @@
1
- export * from "./types";
2
- export * from "./core/matcher";
3
- export * from "./core/filter";
4
- export * from "./core/analyzer";
5
- export * from "./utils/stringUtils";
6
- export * from "./utils/regexUtils";
7
- export * from "./utils/similarityUtils";
8
- export * from "./config/options";
9
-
10
- import { filter, isProfane } from "./core/filter";
1
+ export * from './types';
2
+ export * from './core/matcher';
3
+ export * from './core/filter';
4
+ export * from './core/analyzer';
5
+ export * from './utils/stringUtils';
6
+ export * from './utils/regexUtils';
7
+ export * from './utils/similarityUtils';
8
+ export * from './config/options';
9
+
10
+ import { filter, isProfane } from './core/filter';
11
11
  import {
12
12
  analyze,
13
13
  batchAnalyze,
14
14
  analyzeBySentence,
15
15
  analyzeWithContext,
16
- } from "./core/analyzer";
17
- import { FilterOptions, FilterResult, AnalysisResult } from "./types";
16
+ } from './core/analyzer';
17
+ import { FilterOptions, FilterResult, AnalysisResult } from './types';
18
18
  import {
19
19
  DEFAULT_OPTIONS,
20
20
  FILTER_PRESETS,
21
21
  CATEGORY_PRESETS,
22
22
  REGION_PRESETS,
23
23
  getPresetOptions,
24
- makeRandomGrawlixString,
25
- } from "./config/options";
24
+ } from './config/options';
26
25
 
27
26
  export class IDProfanityFilter {
28
27
  private options: FilterOptions;
@@ -161,10 +160,30 @@ export class IDProfanityFilter {
161
160
  /**
162
161
  * Mengaktifkan deteksi berdasarkan kesamaan
163
162
  * @param threshold Threshold kesamaan (0-1)
163
+ * @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
164
+ * @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
164
165
  */
165
- enableSimilarityDetection(threshold: number = 0.8) {
166
+ enableSimilarityDetection(
167
+ threshold: number = 0.8,
168
+ useLevenshtein: boolean = false,
169
+ maxLevenshteinDistance: number = 2,
170
+ ) {
171
+ this.options.detectSimilarity = true;
172
+ this.options.similarityThreshold = threshold;
173
+ this.options.useLevenshtein = useLevenshtein;
174
+ this.options.maxLevenshteinDistance = maxLevenshteinDistance;
175
+ }
176
+
177
+ /**
178
+ * Mengaktifkan deteksi berbasis Levenshtein distance
179
+ * @param threshold Threshold kesamaan (0-1)
180
+ * @param maxDistance Jarak maksimal Levenshtein (default: 2)
181
+ */
182
+ enableLevenshteinDetection(threshold: number = 0.8, maxDistance: number = 2) {
166
183
  this.options.detectSimilarity = true;
184
+ this.options.useLevenshtein = true;
167
185
  this.options.similarityThreshold = threshold;
186
+ this.options.maxLevenshteinDistance = maxDistance;
168
187
  }
169
188
  }
170
189
 
@@ -50,9 +50,11 @@ export interface FilterOptions {
50
50
  useRandomGrawlix?: boolean;
51
51
  keepFirstAndLast?: boolean;
52
52
  indonesianVariation?: boolean;
53
- detectSimilarity?: boolean;
54
- similarityThreshold?: number;
53
+ detectSimilarity?: boolean; // Enable/disable Levenshtein distance matching
54
+ similarityThreshold?: number; // Threshold for Levenshtein distance similarity (0-1)
55
55
  detectSplit?: boolean;
56
+ useLevenshtein?: boolean; // New option specifically for Levenshtein algorithm
57
+ maxLevenshteinDistance?: number; // Maximum allowed Levenshtein distance
56
58
  }
57
59
 
58
60
  export interface FilterResult {
@@ -72,7 +72,6 @@ export function addLeetSpeakVariations(pattern: string): string {
72
72
  z: ["z", "2"],
73
73
  };
74
74
 
75
- // Ganti tiap karakter dengan variasinya dalam grup character class
76
75
  return pattern
77
76
  .split("")
78
77
  .map((char) => {
@@ -91,6 +91,48 @@ export function findMostSimilar(
91
91
  return mostSimilar;
92
92
  }
93
93
 
94
+ /**
95
+ * Mencari string yang paling mirip dari array menggunakan Levenshtein distance
96
+ *
97
+ * @param target String target
98
+ * @param candidates Array string kandidat
99
+ * @param threshold Minimum kesamaan yang diterima (0-1)
100
+ * @param maxDistance Jarak Levenshtein maksimal yang diterima (default: 3)
101
+ * @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
102
+ */
103
+ export function findMostSimilarWithLevenshtein(
104
+ target: string,
105
+ candidates: string[],
106
+ threshold: number = 0.7,
107
+ maxDistance: number = 3,
108
+ ): string | null {
109
+ if (!candidates.length) return null;
110
+
111
+ let maxSimilarity = 0;
112
+ let minDistance = Infinity;
113
+ let mostSimilar: string | null = null;
114
+
115
+ for (const candidate of candidates) {
116
+ if (Math.abs(target.length - candidate.length) > maxDistance) continue;
117
+
118
+ const distance = levenshteinDistance(target, candidate);
119
+ const similarity = stringSimilarity(target, candidate);
120
+
121
+ if (
122
+ (similarity > maxSimilarity && similarity >= threshold) ||
123
+ (similarity >= threshold && distance < minDistance)
124
+ ) {
125
+ maxSimilarity = similarity;
126
+ minDistance = distance;
127
+ mostSimilar = candidate;
128
+
129
+ if (distance <= 1 || similarity > 0.95) break;
130
+ }
131
+ }
132
+
133
+ return mostSimilar;
134
+ }
135
+
94
136
  /**
95
137
  * Cek apakah string mungkin merupakan variasi dari kata kotor
96
138
  * menggunakan kesamaan string
@@ -170,11 +212,9 @@ export function findPossibleProfanityBySimiliarity(
170
212
  const result: Array<{ word: string; original: string; similarity: number }> =
171
213
  [];
172
214
 
173
- // Pisahkan teks menjadi kata-kata
174
215
  const words = text.toLowerCase().split(/\s+/);
175
216
 
176
217
  for (const word of words) {
177
- // Lewati kata-kata yang terlalu pendek
178
218
  if (word.length < 3) continue;
179
219
 
180
220
  for (const profanity of profanityWords) {
@@ -193,3 +233,58 @@ export function findPossibleProfanityBySimiliarity(
193
233
 
194
234
  return result;
195
235
  }
236
+
237
+ /**
238
+ * Cari kata-kata kotor yang mungkin dari teks menggunakan Levenshtein distance
239
+ *
240
+ * @param text Teks yang akan diperiksa
241
+ * @param profanityWords Daftar kata kotor
242
+ * @param threshold Batas minimum kesamaan (default: 0.8)
243
+ * @param maxDistance Jarak Levenshtein maksimal (default: 2)
244
+ * @returns Array kata yang mungkin merupakan kata kotor
245
+ */
246
+ export function findProfanityByLevenshteinDistance(
247
+ text: string,
248
+ profanityWords: string[],
249
+ threshold: number = 0.8,
250
+ maxDistance: number = 2,
251
+ ): Array<{
252
+ word: string;
253
+ original: string;
254
+ similarity: number;
255
+ distance: number;
256
+ }> {
257
+ const result: Array<{
258
+ word: string;
259
+ original: string;
260
+ similarity: number;
261
+ distance: number;
262
+ }> = [];
263
+
264
+ const words = text.toLowerCase().split(/\s+/);
265
+
266
+ for (const word of words) {
267
+ if (word.length < 3) continue;
268
+
269
+ for (const profanity of profanityWords) {
270
+ if (Math.abs(word.length - profanity.length) > maxDistance) continue;
271
+
272
+ const distance = levenshteinDistance(word, profanity);
273
+ if (distance <= maxDistance) {
274
+ const similarity = stringSimilarity(word, profanity);
275
+
276
+ if (similarity >= threshold) {
277
+ result.push({
278
+ word,
279
+ original: profanity,
280
+ similarity,
281
+ distance,
282
+ });
283
+ break;
284
+ }
285
+ }
286
+ }
287
+ }
288
+
289
+ return result;
290
+ }
package/test.js ADDED
@@ -0,0 +1,184 @@
1
+ // full-example.js
2
+ const { IDProfanityFilter, idFilter } = require('./dist');
3
+
4
+ /**
5
+ * Contoh Penggunaan Lengkap ID-Profanity-Filter
6
+ * ============================================
7
+ * File ini mencakup semua contoh penggunaan utama library.
8
+ */
9
+
10
+ console.log('============ CONTOH PENGGUNAAN DASAR ============\n');
11
+
12
+ const filter = new IDProfanityFilter();
13
+
14
+ const teks =
15
+ 'Dasar kontol kontol kntl babi asu ngentod kamu, jangan banyak bacot! perek';
16
+
17
+ // ===== 1. Cek apakah teks mengandung kata kotor =====
18
+ const hasProfanity = filter.isProfane(teks);
19
+ console.log('1. Apakah teks mengandung kata kotor?', hasProfanity);
20
+
21
+ // ===== 2. Filter kata kotor (mengganti dengan sensor) =====
22
+ const hasil = filter.filter(teks);
23
+ console.log('\n2. Hasil filter:');
24
+ console.log('- Teks asli:', teks);
25
+ console.log('- Teks tersensor:', hasil.filtered);
26
+ console.log('- Jumlah kata yang disensor:', hasil.censored);
27
+ console.log('- Detail penggantian:', hasil.replacements);
28
+
29
+ // ===== 3. Analisis konten =====
30
+ const analisis = filter.analyze(teks);
31
+ console.log('\n3. Hasil analisis:');
32
+ console.log('- Mengandung kata kotor:', analisis.hasProfanity);
33
+ console.log('- Kata kotor yang ditemukan:', analisis.matches);
34
+ console.log('- Kategori kata kotor:', analisis.categories);
35
+ console.log('- Daerah asal kata kotor:', analisis.regions);
36
+ console.log('- Skor keparahan:', analisis.severityScore.toFixed(2));
37
+
38
+ // console.log('\n============ PENGGUNAAN PRESET ============\n');
39
+
40
+ // // ===== 4. Menggunakan preset filter =====
41
+ // console.log('4. Menggunakan preset filter:');
42
+
43
+ // // Preset strict
44
+ // filter.usePreset('strict');
45
+ // console.log('- Preset strict:', filter.filter(teks).filtered);
46
+
47
+ // // Preset childSafe
48
+ // filter.usePreset('childSafe');
49
+ // console.log('- Preset childSafe:', filter.filter(teks).filtered);
50
+
51
+ // // Preset light
52
+ // filter.usePreset('light');
53
+ // console.log('- Preset light:', filter.filter(teks).filtered);
54
+
55
+ console.log('\n============ KUSTOMISASI FILTER ============\n');
56
+
57
+ // ===== 5. Kustomisasi filter =====
58
+ console.log('5. Kustomisasi filter:');
59
+
60
+ filter.setOptions({
61
+ replaceWith: '#', // Menggunakan # sebagai karakter pengganti
62
+ fullWordCensor: false, // Hanya menyensor sebagian kata
63
+ keepFirstAndLast: true, // Menyimpan huruf pertama dan terakhir
64
+ });
65
+
66
+ console.log('- Filter dengan opsi kustom:', filter.filter(teks).filtered);
67
+
68
+ // Mendeteksi variasi penulisan
69
+ filter.setOptions({
70
+ detectLeetSpeak: true,
71
+ indonesianVariation: true,
72
+ detectSplit: true,
73
+ detectSimilarity: true,
74
+ useLevenshtein: true,
75
+ similarityThreshold: 0.85,
76
+ maxLevenshteinDistance: 2,
77
+ });
78
+
79
+ const teks2 =
80
+ 'Dasar b4b1 kamu, j4nc0k! Ngent0d! k-o-n-t-o-l! k o n t o l k-0nt0l! anjiing kontool kwontol';
81
+ console.log('- Teks asli dengan variasi:', teks2);
82
+ console.log('- Hasil filter variasi:', filter.filter(teks2).filtered);
83
+
84
+ // ===== 6. Menggunakan whitelist =====
85
+ console.log('\n6. Menggunakan whitelist:');
86
+
87
+ filter.addToWhitelist('anjing');
88
+ const tekstBinatang =
89
+ 'Anjing itu hewan peliharaan yang setia, tidak seperti bajingan itu';
90
+
91
+ console.log('- Teks dengan "anjing" dalam konteks binatang:', tekstBinatang);
92
+ console.log(
93
+ '- Hasil filter dengan whitelist:',
94
+ filter.filter(tekstBinatang).filtered,
95
+ );
96
+
97
+ // ===== 7. Menggunakan daftar kata kustom =====
98
+ console.log('\n7. Menggunakan daftar kata kustom:');
99
+
100
+ const customBadWords = ['jelek', 'buruk', 'sampah', 'payah'];
101
+
102
+ // Reset filter dan gunakan daftar kustom
103
+ filter.setOptions({
104
+ wordList: customBadWords,
105
+ replaceWith: '*',
106
+ });
107
+
108
+ const teksKustom = 'Film ini jelek dan payah sekali!';
109
+ console.log('- Teks asli:', teksKustom);
110
+ console.log('- Hasil filter kustom:', filter.filter(teksKustom).filtered);
111
+
112
+ console.log('\n============ ANALISIS LANJUTAN ============\n');
113
+
114
+ // ===== 8. Analisis per kalimat =====
115
+ console.log('8. Analisis per kalimat:');
116
+
117
+ filter.setOptions({}); // Reset ke default
118
+ const kalimat =
119
+ 'Saya sangat suka filmnya. Tapi pemainnya seperti anjing, aktingnya buruk.';
120
+ console.log('- Kalimat:', kalimat);
121
+
122
+ const kalimatAnalisis = filter.analyzeBySentence(kalimat);
123
+ console.log('- Hasil analisis per kalimat:');
124
+ kalimatAnalisis.forEach((hasil, index) => {
125
+ console.log(` Kalimat ${index + 1}: "${hasil.sentence}"`);
126
+ console.log(` Mengandung kata kotor: ${hasil.hasProfanity}`);
127
+ if (hasil.hasProfanity) {
128
+ console.log(` Kata kotor: ${hasil.matches.join(', ')}`);
129
+ }
130
+ });
131
+
132
+ // ===== 9. Analisis batch untuk komentar =====
133
+ console.log('\n9. Analisis batch:');
134
+
135
+ const komentar = [
136
+ 'Film ini sangat bagus, ceritanya menarik sekali!',
137
+ 'Dasar goblok, sialan kamu!',
138
+ 'Anjing emang filmnya, sampah banget.',
139
+ ];
140
+
141
+ console.log('- Batch teks:');
142
+ komentar.forEach((k, i) => console.log(` ${i + 1}. "${k}"`));
143
+
144
+ const hasilBatch = filter.batchAnalyze(komentar);
145
+ console.log('- Hasil analisis batch:');
146
+ console.log(` - Total teks: ${hasilBatch.totalTexts}`);
147
+ console.log(` - Teks mengandung kata kotor: ${hasilBatch.profaneTexts}`);
148
+ console.log(` - Teks bersih: ${hasilBatch.cleanTexts}`);
149
+ console.log(` - Kategori teratas: ${hasilBatch.topCategories.join(', ')}`);
150
+ console.log(` - Daerah teratas: ${hasilBatch.topRegions.join(', ')}`);
151
+ console.log(' - Kata kotor terbanyak:');
152
+ hasilBatch.mostFrequentWords.forEach((word) => {
153
+ console.log(` * "${word.word}": ${word.count} kali`);
154
+ });
155
+
156
+ // ===== 10. Filter berdasarkan kategori dan daerah =====
157
+ console.log('\n10. Filter berdasarkan kategori dan daerah:');
158
+
159
+ // Filter untuk kata dari daerah Jawa saja
160
+ const filterJawa = new IDProfanityFilter({
161
+ regions: ['jawa'], // Hanya filter kata dari Jawa
162
+ });
163
+
164
+ const contohTeksJawa = 'Dasar jancok, asu, bangsat, bacot!';
165
+ console.log('- Teks asli:', contohTeksJawa);
166
+ console.log(
167
+ '- Filter khusus kata Jawa:',
168
+ filterJawa.filter(contohTeksJawa).filtered,
169
+ );
170
+
171
+ // Filter untuk kategori sexual saja
172
+ const filterSexual = new IDProfanityFilter({
173
+ categories: ['sexual'], // Hanya filter kata kategori seksual
174
+ });
175
+
176
+ const contohTeksSexual =
177
+ 'Kata kotor seperti anjing dan goblok tidak disensor, tapi bokep disensor.';
178
+ console.log('- Teks asli:', contohTeksSexual);
179
+ console.log(
180
+ '- Filter khusus kategori sexual:',
181
+ filterSexual.filter(contohTeksSexual).filtered,
182
+ );
183
+
184
+ console.log('\n============ SELESAI ============');
@@ -1,31 +0,0 @@
1
- // import { sexual } from './sexual';
2
- // import { insult } from './insult';
3
- // import { profanity } from './profanity';
4
- // import { slur } from './slur';
5
- // import { drugs } from './drugs';
6
- // import { disgusting } from './disgusting';
7
- // import { blasphemy } from './blasphemy';
8
-
9
- // export const categories = {
10
- // sexual, // Kata-kata berbau seksual
11
- // insult, // Kata-kata penghinaan
12
- // profanity, // Umpatan umum
13
- // slur, // Perkataan merendahkan berdasarkan identitas
14
- // drugs, // Terkait narkoba
15
- // disgusting, // Kata-kata menjijikkan
16
- // blasphemy, // Penistaan agama
17
- // };
18
-
19
- // export { sexual, insult, profanity, slur, drugs, disgusting, blasphemy };
20
-
21
- // export const allCategoryWords = [
22
- // ...sexual,
23
- // ...insult,
24
- // ...profanity,
25
- // ...slur,
26
- // ...drugs,
27
- // ...disgusting,
28
- // ...blasphemy,
29
- // ];
30
-
31
- // export default allCategoryWords;