@sideid/id-profanity-filter 1.9.5 → 1.10.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.eslintrc.js +30 -2
- package/README.md +272 -9
- package/dist/config/options.d.ts +24 -0
- package/dist/constants/categories/index.d.ts +9 -0
- package/dist/constants/regions/index.d.ts +8 -0
- package/dist/constants/wordList.d.ts +1 -1
- package/dist/core/analyzer.d.ts +1 -1
- package/dist/core/filter.d.ts +1 -1
- package/dist/core/matcher.d.ts +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.esm.js +411 -77
- package/dist/index.esm.js.map +1 -1
- package/dist/index.js +412 -76
- package/dist/index.js.map +1 -1
- package/dist/types/index.d.ts +2 -0
- package/dist/utils/similarityUtils.d.ts +25 -0
- package/eslint.config.mjs +40 -0
- package/examples/advanced.ts +120 -0
- package/examples/basic.ts +60 -41
- package/examples/custom-list.ts +140 -0
- package/package.json +2 -2
- package/src/config/options.ts +2 -0
- package/src/constants/regions/general.ts +102 -2
- package/src/constants/regions/jawa.ts +10 -0
- package/src/constants/wordList.ts +21 -14
- package/src/core/analyzer.ts +8 -14
- package/src/core/filter.ts +182 -41
- package/src/core/matcher.ts +131 -61
- package/src/index.ts +34 -15
- package/src/types/index.ts +4 -2
- package/src/utils/regexUtils.ts +0 -1
- package/src/utils/similarityUtils.ts +97 -2
- package/test.js +184 -0
- package/src/constants/categories/index.ts +0 -31
- package/src/constants/regions/index.ts +0 -62
package/src/core/matcher.ts
CHANGED
|
@@ -3,26 +3,21 @@ import {
|
|
|
3
3
|
ProfanityCategory,
|
|
4
4
|
Region,
|
|
5
5
|
FilterOptions,
|
|
6
|
-
} from
|
|
6
|
+
} from '../types';
|
|
7
7
|
|
|
8
|
-
import { wordObjects
|
|
9
|
-
import {
|
|
10
|
-
|
|
11
|
-
escapeRegExp,
|
|
12
|
-
containsAnyWord,
|
|
13
|
-
detectSplitWords,
|
|
14
|
-
} from "../utils/stringUtils";
|
|
15
|
-
import {
|
|
16
|
-
createWordRegex,
|
|
17
|
-
addLeetSpeakVariations,
|
|
18
|
-
addIndonesianVariations,
|
|
19
|
-
addSplitVariations,
|
|
20
|
-
} from "../utils/regexUtils";
|
|
8
|
+
import { wordObjects } from '../constants/wordList';
|
|
9
|
+
import { normalizeText } from '../utils/stringUtils';
|
|
10
|
+
import { createWordRegex } from '../utils/regexUtils';
|
|
21
11
|
import {
|
|
22
12
|
findPossibleProfanityBySimiliarity,
|
|
23
|
-
|
|
24
|
-
} from
|
|
25
|
-
import { DEFAULT_OPTIONS } from
|
|
13
|
+
findProfanityByLevenshteinDistance,
|
|
14
|
+
} from '../utils/similarityUtils';
|
|
15
|
+
import { DEFAULT_OPTIONS } from '../config/options';
|
|
16
|
+
|
|
17
|
+
interface FindProfanityFunction {
|
|
18
|
+
(text: string, options?: FilterOptions): string[];
|
|
19
|
+
lastActualMatches?: Map<string, string[]>;
|
|
20
|
+
}
|
|
26
21
|
|
|
27
22
|
export function findProfanity(
|
|
28
23
|
text: string,
|
|
@@ -40,38 +35,55 @@ export function findProfanity(
|
|
|
40
35
|
detectSimilarity = false,
|
|
41
36
|
similarityThreshold = 0.8,
|
|
42
37
|
detectSplit = false,
|
|
38
|
+
useLevenshtein = false,
|
|
39
|
+
maxLevenshteinDistance = 2,
|
|
43
40
|
} = { ...DEFAULT_OPTIONS, ...options };
|
|
44
41
|
|
|
45
42
|
const normalizedText = normalizeText(text);
|
|
46
43
|
|
|
47
|
-
let
|
|
44
|
+
let baseWordsToCheck: string[] = wordList.length > 0 ? wordList : [];
|
|
48
45
|
|
|
49
|
-
if (
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
.
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
.map((word) => word.word);
|
|
61
|
-
} else {
|
|
62
|
-
wordsToCheck = wordObjects.map((word) => word.word);
|
|
63
|
-
}
|
|
46
|
+
if (baseWordsToCheck.length === 0) {
|
|
47
|
+
const filteredWords = wordObjects.filter((word) => {
|
|
48
|
+
const matchCategory = categories
|
|
49
|
+
? categories.includes(word.category)
|
|
50
|
+
: true;
|
|
51
|
+
const matchRegion = regions ? regions.includes(word.region) : true;
|
|
52
|
+
const matchSeverity = word.severity >= severityThreshold;
|
|
53
|
+
return matchCategory && matchRegion && matchSeverity;
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
baseWordsToCheck = filteredWords.map((word) => word.word);
|
|
64
57
|
}
|
|
65
58
|
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
59
|
+
const aliasMap = new Map<string, string>();
|
|
60
|
+
wordObjects.forEach((wordObj) => {
|
|
61
|
+
if (wordObj.aliases && wordObj.aliases.length > 0) {
|
|
62
|
+
const matchCategory = categories
|
|
63
|
+
? categories.includes(wordObj.category)
|
|
64
|
+
: true;
|
|
65
|
+
const matchRegion = regions ? regions.includes(wordObj.region) : true;
|
|
66
|
+
const matchSeverity = wordObj.severity >= severityThreshold;
|
|
67
|
+
|
|
68
|
+
if (matchCategory && matchRegion && matchSeverity) {
|
|
69
|
+
wordObj.aliases.forEach((alias) => {
|
|
70
|
+
aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
|
|
71
|
+
});
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
const wordsToCheck = [
|
|
77
|
+
...baseWordsToCheck,
|
|
78
|
+
...Array.from(aliasMap.keys()),
|
|
79
|
+
].filter((word) => !whitelist.includes(word.toLowerCase()));
|
|
69
80
|
|
|
70
81
|
if (wordsToCheck.length === 0) {
|
|
71
82
|
return [];
|
|
72
83
|
}
|
|
73
84
|
|
|
74
85
|
const matches = new Set<string>();
|
|
86
|
+
const actualMatches = new Map<string, string[]>();
|
|
75
87
|
|
|
76
88
|
wordsToCheck.forEach((word) => {
|
|
77
89
|
const regex = createWordRegex(word, {
|
|
@@ -84,7 +96,14 @@ export function findProfanity(
|
|
|
84
96
|
|
|
85
97
|
let match;
|
|
86
98
|
while ((match = regex.exec(normalizedText)) !== null) {
|
|
87
|
-
|
|
99
|
+
const originalWord =
|
|
100
|
+
aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
101
|
+
matches.add(originalWord);
|
|
102
|
+
|
|
103
|
+
if (!actualMatches.has(originalWord)) {
|
|
104
|
+
actualMatches.set(originalWord, []);
|
|
105
|
+
}
|
|
106
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
88
107
|
}
|
|
89
108
|
});
|
|
90
109
|
|
|
@@ -100,7 +119,14 @@ export function findProfanity(
|
|
|
100
119
|
|
|
101
120
|
let match;
|
|
102
121
|
while ((match = leetRegex.exec(text)) !== null) {
|
|
103
|
-
|
|
122
|
+
const originalWord =
|
|
123
|
+
aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
124
|
+
matches.add(originalWord);
|
|
125
|
+
|
|
126
|
+
if (!actualMatches.has(originalWord)) {
|
|
127
|
+
actualMatches.set(originalWord, []);
|
|
128
|
+
}
|
|
129
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
104
130
|
}
|
|
105
131
|
});
|
|
106
132
|
}
|
|
@@ -117,41 +143,85 @@ export function findProfanity(
|
|
|
117
143
|
|
|
118
144
|
let match;
|
|
119
145
|
while ((match = variantRegex.exec(text)) !== null) {
|
|
120
|
-
|
|
146
|
+
const originalWord =
|
|
147
|
+
aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
148
|
+
matches.add(originalWord);
|
|
149
|
+
|
|
150
|
+
if (!actualMatches.has(originalWord)) {
|
|
151
|
+
actualMatches.set(originalWord, []);
|
|
152
|
+
}
|
|
153
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
121
154
|
}
|
|
122
155
|
});
|
|
123
156
|
}
|
|
124
157
|
|
|
125
158
|
if (detectSplit) {
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
});
|
|
159
|
+
wordsToCheck.forEach((word) => {
|
|
160
|
+
const splitRegex = createWordRegex(word, {
|
|
161
|
+
wholeWord: false,
|
|
162
|
+
caseSensitive: false,
|
|
163
|
+
leetSpeak: false,
|
|
164
|
+
detectSplit: true,
|
|
165
|
+
indonesianVariation: false,
|
|
166
|
+
});
|
|
135
167
|
|
|
136
|
-
|
|
137
|
-
|
|
168
|
+
let match;
|
|
169
|
+
while ((match = splitRegex.exec(text)) !== null) {
|
|
170
|
+
const originalWord =
|
|
171
|
+
aliasMap.get(word.toLowerCase()) || word.toLowerCase();
|
|
172
|
+
matches.add(originalWord);
|
|
173
|
+
|
|
174
|
+
if (!actualMatches.has(originalWord)) {
|
|
175
|
+
actualMatches.set(originalWord, []);
|
|
138
176
|
}
|
|
139
|
-
|
|
140
|
-
|
|
177
|
+
actualMatches.get(originalWord)?.push(match[0]);
|
|
178
|
+
}
|
|
179
|
+
});
|
|
141
180
|
}
|
|
142
181
|
|
|
143
182
|
if (detectSimilarity) {
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
183
|
+
if (useLevenshtein) {
|
|
184
|
+
const possibleProfanity = findProfanityByLevenshteinDistance(
|
|
185
|
+
text,
|
|
186
|
+
wordsToCheck,
|
|
187
|
+
similarityThreshold,
|
|
188
|
+
maxLevenshteinDistance,
|
|
189
|
+
);
|
|
190
|
+
|
|
191
|
+
possibleProfanity.forEach((item) => {
|
|
192
|
+
const originalWord =
|
|
193
|
+
aliasMap.get(item.original.toLowerCase()) ||
|
|
194
|
+
item.original.toLowerCase();
|
|
195
|
+
matches.add(originalWord);
|
|
196
|
+
|
|
197
|
+
if (!actualMatches.has(originalWord)) {
|
|
198
|
+
actualMatches.set(originalWord, []);
|
|
199
|
+
}
|
|
200
|
+
actualMatches.get(originalWord)?.push(item.word);
|
|
201
|
+
});
|
|
202
|
+
} else {
|
|
203
|
+
const possibleProfanity = findPossibleProfanityBySimiliarity(
|
|
204
|
+
text,
|
|
205
|
+
wordsToCheck,
|
|
206
|
+
similarityThreshold,
|
|
207
|
+
);
|
|
208
|
+
|
|
209
|
+
possibleProfanity.forEach((item) => {
|
|
210
|
+
matches.add(item.original.toLowerCase());
|
|
211
|
+
|
|
212
|
+
const originalWord =
|
|
213
|
+
aliasMap.get(item.original.toLowerCase()) ||
|
|
214
|
+
item.original.toLowerCase();
|
|
215
|
+
if (!actualMatches.has(originalWord)) {
|
|
216
|
+
actualMatches.set(originalWord, []);
|
|
217
|
+
}
|
|
218
|
+
actualMatches.get(originalWord)?.push(item.word);
|
|
219
|
+
});
|
|
220
|
+
}
|
|
153
221
|
}
|
|
154
222
|
|
|
223
|
+
(findProfanity as FindProfanityFunction).lastActualMatches = actualMatches;
|
|
224
|
+
|
|
155
225
|
return Array.from(matches);
|
|
156
226
|
}
|
|
157
227
|
|
package/src/index.ts
CHANGED
|
@@ -1,28 +1,27 @@
|
|
|
1
|
-
export * from
|
|
2
|
-
export * from
|
|
3
|
-
export * from
|
|
4
|
-
export * from
|
|
5
|
-
export * from
|
|
6
|
-
export * from
|
|
7
|
-
export * from
|
|
8
|
-
export * from
|
|
9
|
-
|
|
10
|
-
import { filter, isProfane } from
|
|
1
|
+
export * from './types';
|
|
2
|
+
export * from './core/matcher';
|
|
3
|
+
export * from './core/filter';
|
|
4
|
+
export * from './core/analyzer';
|
|
5
|
+
export * from './utils/stringUtils';
|
|
6
|
+
export * from './utils/regexUtils';
|
|
7
|
+
export * from './utils/similarityUtils';
|
|
8
|
+
export * from './config/options';
|
|
9
|
+
|
|
10
|
+
import { filter, isProfane } from './core/filter';
|
|
11
11
|
import {
|
|
12
12
|
analyze,
|
|
13
13
|
batchAnalyze,
|
|
14
14
|
analyzeBySentence,
|
|
15
15
|
analyzeWithContext,
|
|
16
|
-
} from
|
|
17
|
-
import { FilterOptions, FilterResult, AnalysisResult } from
|
|
16
|
+
} from './core/analyzer';
|
|
17
|
+
import { FilterOptions, FilterResult, AnalysisResult } from './types';
|
|
18
18
|
import {
|
|
19
19
|
DEFAULT_OPTIONS,
|
|
20
20
|
FILTER_PRESETS,
|
|
21
21
|
CATEGORY_PRESETS,
|
|
22
22
|
REGION_PRESETS,
|
|
23
23
|
getPresetOptions,
|
|
24
|
-
|
|
25
|
-
} from "./config/options";
|
|
24
|
+
} from './config/options';
|
|
26
25
|
|
|
27
26
|
export class IDProfanityFilter {
|
|
28
27
|
private options: FilterOptions;
|
|
@@ -161,10 +160,30 @@ export class IDProfanityFilter {
|
|
|
161
160
|
/**
|
|
162
161
|
* Mengaktifkan deteksi berdasarkan kesamaan
|
|
163
162
|
* @param threshold Threshold kesamaan (0-1)
|
|
163
|
+
* @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
|
|
164
|
+
* @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
|
|
164
165
|
*/
|
|
165
|
-
enableSimilarityDetection(
|
|
166
|
+
enableSimilarityDetection(
|
|
167
|
+
threshold: number = 0.8,
|
|
168
|
+
useLevenshtein: boolean = false,
|
|
169
|
+
maxLevenshteinDistance: number = 2,
|
|
170
|
+
) {
|
|
171
|
+
this.options.detectSimilarity = true;
|
|
172
|
+
this.options.similarityThreshold = threshold;
|
|
173
|
+
this.options.useLevenshtein = useLevenshtein;
|
|
174
|
+
this.options.maxLevenshteinDistance = maxLevenshteinDistance;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* Mengaktifkan deteksi berbasis Levenshtein distance
|
|
179
|
+
* @param threshold Threshold kesamaan (0-1)
|
|
180
|
+
* @param maxDistance Jarak maksimal Levenshtein (default: 2)
|
|
181
|
+
*/
|
|
182
|
+
enableLevenshteinDetection(threshold: number = 0.8, maxDistance: number = 2) {
|
|
166
183
|
this.options.detectSimilarity = true;
|
|
184
|
+
this.options.useLevenshtein = true;
|
|
167
185
|
this.options.similarityThreshold = threshold;
|
|
186
|
+
this.options.maxLevenshteinDistance = maxDistance;
|
|
168
187
|
}
|
|
169
188
|
}
|
|
170
189
|
|
package/src/types/index.ts
CHANGED
|
@@ -50,9 +50,11 @@ export interface FilterOptions {
|
|
|
50
50
|
useRandomGrawlix?: boolean;
|
|
51
51
|
keepFirstAndLast?: boolean;
|
|
52
52
|
indonesianVariation?: boolean;
|
|
53
|
-
detectSimilarity?: boolean;
|
|
54
|
-
similarityThreshold?: number;
|
|
53
|
+
detectSimilarity?: boolean; // Enable/disable Levenshtein distance matching
|
|
54
|
+
similarityThreshold?: number; // Threshold for Levenshtein distance similarity (0-1)
|
|
55
55
|
detectSplit?: boolean;
|
|
56
|
+
useLevenshtein?: boolean; // New option specifically for Levenshtein algorithm
|
|
57
|
+
maxLevenshteinDistance?: number; // Maximum allowed Levenshtein distance
|
|
56
58
|
}
|
|
57
59
|
|
|
58
60
|
export interface FilterResult {
|
package/src/utils/regexUtils.ts
CHANGED
|
@@ -91,6 +91,48 @@ export function findMostSimilar(
|
|
|
91
91
|
return mostSimilar;
|
|
92
92
|
}
|
|
93
93
|
|
|
94
|
+
/**
|
|
95
|
+
* Mencari string yang paling mirip dari array menggunakan Levenshtein distance
|
|
96
|
+
*
|
|
97
|
+
* @param target String target
|
|
98
|
+
* @param candidates Array string kandidat
|
|
99
|
+
* @param threshold Minimum kesamaan yang diterima (0-1)
|
|
100
|
+
* @param maxDistance Jarak Levenshtein maksimal yang diterima (default: 3)
|
|
101
|
+
* @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
|
|
102
|
+
*/
|
|
103
|
+
export function findMostSimilarWithLevenshtein(
|
|
104
|
+
target: string,
|
|
105
|
+
candidates: string[],
|
|
106
|
+
threshold: number = 0.7,
|
|
107
|
+
maxDistance: number = 3,
|
|
108
|
+
): string | null {
|
|
109
|
+
if (!candidates.length) return null;
|
|
110
|
+
|
|
111
|
+
let maxSimilarity = 0;
|
|
112
|
+
let minDistance = Infinity;
|
|
113
|
+
let mostSimilar: string | null = null;
|
|
114
|
+
|
|
115
|
+
for (const candidate of candidates) {
|
|
116
|
+
if (Math.abs(target.length - candidate.length) > maxDistance) continue;
|
|
117
|
+
|
|
118
|
+
const distance = levenshteinDistance(target, candidate);
|
|
119
|
+
const similarity = stringSimilarity(target, candidate);
|
|
120
|
+
|
|
121
|
+
if (
|
|
122
|
+
(similarity > maxSimilarity && similarity >= threshold) ||
|
|
123
|
+
(similarity >= threshold && distance < minDistance)
|
|
124
|
+
) {
|
|
125
|
+
maxSimilarity = similarity;
|
|
126
|
+
minDistance = distance;
|
|
127
|
+
mostSimilar = candidate;
|
|
128
|
+
|
|
129
|
+
if (distance <= 1 || similarity > 0.95) break;
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
return mostSimilar;
|
|
134
|
+
}
|
|
135
|
+
|
|
94
136
|
/**
|
|
95
137
|
* Cek apakah string mungkin merupakan variasi dari kata kotor
|
|
96
138
|
* menggunakan kesamaan string
|
|
@@ -170,11 +212,9 @@ export function findPossibleProfanityBySimiliarity(
|
|
|
170
212
|
const result: Array<{ word: string; original: string; similarity: number }> =
|
|
171
213
|
[];
|
|
172
214
|
|
|
173
|
-
// Pisahkan teks menjadi kata-kata
|
|
174
215
|
const words = text.toLowerCase().split(/\s+/);
|
|
175
216
|
|
|
176
217
|
for (const word of words) {
|
|
177
|
-
// Lewati kata-kata yang terlalu pendek
|
|
178
218
|
if (word.length < 3) continue;
|
|
179
219
|
|
|
180
220
|
for (const profanity of profanityWords) {
|
|
@@ -193,3 +233,58 @@ export function findPossibleProfanityBySimiliarity(
|
|
|
193
233
|
|
|
194
234
|
return result;
|
|
195
235
|
}
|
|
236
|
+
|
|
237
|
+
/**
|
|
238
|
+
* Cari kata-kata kotor yang mungkin dari teks menggunakan Levenshtein distance
|
|
239
|
+
*
|
|
240
|
+
* @param text Teks yang akan diperiksa
|
|
241
|
+
* @param profanityWords Daftar kata kotor
|
|
242
|
+
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
243
|
+
* @param maxDistance Jarak Levenshtein maksimal (default: 2)
|
|
244
|
+
* @returns Array kata yang mungkin merupakan kata kotor
|
|
245
|
+
*/
|
|
246
|
+
export function findProfanityByLevenshteinDistance(
|
|
247
|
+
text: string,
|
|
248
|
+
profanityWords: string[],
|
|
249
|
+
threshold: number = 0.8,
|
|
250
|
+
maxDistance: number = 2,
|
|
251
|
+
): Array<{
|
|
252
|
+
word: string;
|
|
253
|
+
original: string;
|
|
254
|
+
similarity: number;
|
|
255
|
+
distance: number;
|
|
256
|
+
}> {
|
|
257
|
+
const result: Array<{
|
|
258
|
+
word: string;
|
|
259
|
+
original: string;
|
|
260
|
+
similarity: number;
|
|
261
|
+
distance: number;
|
|
262
|
+
}> = [];
|
|
263
|
+
|
|
264
|
+
const words = text.toLowerCase().split(/\s+/);
|
|
265
|
+
|
|
266
|
+
for (const word of words) {
|
|
267
|
+
if (word.length < 3) continue;
|
|
268
|
+
|
|
269
|
+
for (const profanity of profanityWords) {
|
|
270
|
+
if (Math.abs(word.length - profanity.length) > maxDistance) continue;
|
|
271
|
+
|
|
272
|
+
const distance = levenshteinDistance(word, profanity);
|
|
273
|
+
if (distance <= maxDistance) {
|
|
274
|
+
const similarity = stringSimilarity(word, profanity);
|
|
275
|
+
|
|
276
|
+
if (similarity >= threshold) {
|
|
277
|
+
result.push({
|
|
278
|
+
word,
|
|
279
|
+
original: profanity,
|
|
280
|
+
similarity,
|
|
281
|
+
distance,
|
|
282
|
+
});
|
|
283
|
+
break;
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
return result;
|
|
290
|
+
}
|
package/test.js
ADDED
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
// full-example.js
|
|
2
|
+
const { IDProfanityFilter, idFilter } = require('./dist');
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Contoh Penggunaan Lengkap ID-Profanity-Filter
|
|
6
|
+
* ============================================
|
|
7
|
+
* File ini mencakup semua contoh penggunaan utama library.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
console.log('============ CONTOH PENGGUNAAN DASAR ============\n');
|
|
11
|
+
|
|
12
|
+
const filter = new IDProfanityFilter();
|
|
13
|
+
|
|
14
|
+
const teks =
|
|
15
|
+
'Dasar kontol kontol kntl babi asu ngentod kamu, jangan banyak bacot! perek';
|
|
16
|
+
|
|
17
|
+
// ===== 1. Cek apakah teks mengandung kata kotor =====
|
|
18
|
+
const hasProfanity = filter.isProfane(teks);
|
|
19
|
+
console.log('1. Apakah teks mengandung kata kotor?', hasProfanity);
|
|
20
|
+
|
|
21
|
+
// ===== 2. Filter kata kotor (mengganti dengan sensor) =====
|
|
22
|
+
const hasil = filter.filter(teks);
|
|
23
|
+
console.log('\n2. Hasil filter:');
|
|
24
|
+
console.log('- Teks asli:', teks);
|
|
25
|
+
console.log('- Teks tersensor:', hasil.filtered);
|
|
26
|
+
console.log('- Jumlah kata yang disensor:', hasil.censored);
|
|
27
|
+
console.log('- Detail penggantian:', hasil.replacements);
|
|
28
|
+
|
|
29
|
+
// ===== 3. Analisis konten =====
|
|
30
|
+
const analisis = filter.analyze(teks);
|
|
31
|
+
console.log('\n3. Hasil analisis:');
|
|
32
|
+
console.log('- Mengandung kata kotor:', analisis.hasProfanity);
|
|
33
|
+
console.log('- Kata kotor yang ditemukan:', analisis.matches);
|
|
34
|
+
console.log('- Kategori kata kotor:', analisis.categories);
|
|
35
|
+
console.log('- Daerah asal kata kotor:', analisis.regions);
|
|
36
|
+
console.log('- Skor keparahan:', analisis.severityScore.toFixed(2));
|
|
37
|
+
|
|
38
|
+
// console.log('\n============ PENGGUNAAN PRESET ============\n');
|
|
39
|
+
|
|
40
|
+
// // ===== 4. Menggunakan preset filter =====
|
|
41
|
+
// console.log('4. Menggunakan preset filter:');
|
|
42
|
+
|
|
43
|
+
// // Preset strict
|
|
44
|
+
// filter.usePreset('strict');
|
|
45
|
+
// console.log('- Preset strict:', filter.filter(teks).filtered);
|
|
46
|
+
|
|
47
|
+
// // Preset childSafe
|
|
48
|
+
// filter.usePreset('childSafe');
|
|
49
|
+
// console.log('- Preset childSafe:', filter.filter(teks).filtered);
|
|
50
|
+
|
|
51
|
+
// // Preset light
|
|
52
|
+
// filter.usePreset('light');
|
|
53
|
+
// console.log('- Preset light:', filter.filter(teks).filtered);
|
|
54
|
+
|
|
55
|
+
console.log('\n============ KUSTOMISASI FILTER ============\n');
|
|
56
|
+
|
|
57
|
+
// ===== 5. Kustomisasi filter =====
|
|
58
|
+
console.log('5. Kustomisasi filter:');
|
|
59
|
+
|
|
60
|
+
filter.setOptions({
|
|
61
|
+
replaceWith: '#', // Menggunakan # sebagai karakter pengganti
|
|
62
|
+
fullWordCensor: false, // Hanya menyensor sebagian kata
|
|
63
|
+
keepFirstAndLast: true, // Menyimpan huruf pertama dan terakhir
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
console.log('- Filter dengan opsi kustom:', filter.filter(teks).filtered);
|
|
67
|
+
|
|
68
|
+
// Mendeteksi variasi penulisan
|
|
69
|
+
filter.setOptions({
|
|
70
|
+
detectLeetSpeak: true,
|
|
71
|
+
indonesianVariation: true,
|
|
72
|
+
detectSplit: true,
|
|
73
|
+
detectSimilarity: true,
|
|
74
|
+
useLevenshtein: true,
|
|
75
|
+
similarityThreshold: 0.85,
|
|
76
|
+
maxLevenshteinDistance: 2,
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
const teks2 =
|
|
80
|
+
'Dasar b4b1 kamu, j4nc0k! Ngent0d! k-o-n-t-o-l! k o n t o l k-0nt0l! anjiing kontool kwontol';
|
|
81
|
+
console.log('- Teks asli dengan variasi:', teks2);
|
|
82
|
+
console.log('- Hasil filter variasi:', filter.filter(teks2).filtered);
|
|
83
|
+
|
|
84
|
+
// ===== 6. Menggunakan whitelist =====
|
|
85
|
+
console.log('\n6. Menggunakan whitelist:');
|
|
86
|
+
|
|
87
|
+
filter.addToWhitelist('anjing');
|
|
88
|
+
const tekstBinatang =
|
|
89
|
+
'Anjing itu hewan peliharaan yang setia, tidak seperti bajingan itu';
|
|
90
|
+
|
|
91
|
+
console.log('- Teks dengan "anjing" dalam konteks binatang:', tekstBinatang);
|
|
92
|
+
console.log(
|
|
93
|
+
'- Hasil filter dengan whitelist:',
|
|
94
|
+
filter.filter(tekstBinatang).filtered,
|
|
95
|
+
);
|
|
96
|
+
|
|
97
|
+
// ===== 7. Menggunakan daftar kata kustom =====
|
|
98
|
+
console.log('\n7. Menggunakan daftar kata kustom:');
|
|
99
|
+
|
|
100
|
+
const customBadWords = ['jelek', 'buruk', 'sampah', 'payah'];
|
|
101
|
+
|
|
102
|
+
// Reset filter dan gunakan daftar kustom
|
|
103
|
+
filter.setOptions({
|
|
104
|
+
wordList: customBadWords,
|
|
105
|
+
replaceWith: '*',
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
const teksKustom = 'Film ini jelek dan payah sekali!';
|
|
109
|
+
console.log('- Teks asli:', teksKustom);
|
|
110
|
+
console.log('- Hasil filter kustom:', filter.filter(teksKustom).filtered);
|
|
111
|
+
|
|
112
|
+
console.log('\n============ ANALISIS LANJUTAN ============\n');
|
|
113
|
+
|
|
114
|
+
// ===== 8. Analisis per kalimat =====
|
|
115
|
+
console.log('8. Analisis per kalimat:');
|
|
116
|
+
|
|
117
|
+
filter.setOptions({}); // Reset ke default
|
|
118
|
+
const kalimat =
|
|
119
|
+
'Saya sangat suka filmnya. Tapi pemainnya seperti anjing, aktingnya buruk.';
|
|
120
|
+
console.log('- Kalimat:', kalimat);
|
|
121
|
+
|
|
122
|
+
const kalimatAnalisis = filter.analyzeBySentence(kalimat);
|
|
123
|
+
console.log('- Hasil analisis per kalimat:');
|
|
124
|
+
kalimatAnalisis.forEach((hasil, index) => {
|
|
125
|
+
console.log(` Kalimat ${index + 1}: "${hasil.sentence}"`);
|
|
126
|
+
console.log(` Mengandung kata kotor: ${hasil.hasProfanity}`);
|
|
127
|
+
if (hasil.hasProfanity) {
|
|
128
|
+
console.log(` Kata kotor: ${hasil.matches.join(', ')}`);
|
|
129
|
+
}
|
|
130
|
+
});
|
|
131
|
+
|
|
132
|
+
// ===== 9. Analisis batch untuk komentar =====
|
|
133
|
+
console.log('\n9. Analisis batch:');
|
|
134
|
+
|
|
135
|
+
const komentar = [
|
|
136
|
+
'Film ini sangat bagus, ceritanya menarik sekali!',
|
|
137
|
+
'Dasar goblok, sialan kamu!',
|
|
138
|
+
'Anjing emang filmnya, sampah banget.',
|
|
139
|
+
];
|
|
140
|
+
|
|
141
|
+
console.log('- Batch teks:');
|
|
142
|
+
komentar.forEach((k, i) => console.log(` ${i + 1}. "${k}"`));
|
|
143
|
+
|
|
144
|
+
const hasilBatch = filter.batchAnalyze(komentar);
|
|
145
|
+
console.log('- Hasil analisis batch:');
|
|
146
|
+
console.log(` - Total teks: ${hasilBatch.totalTexts}`);
|
|
147
|
+
console.log(` - Teks mengandung kata kotor: ${hasilBatch.profaneTexts}`);
|
|
148
|
+
console.log(` - Teks bersih: ${hasilBatch.cleanTexts}`);
|
|
149
|
+
console.log(` - Kategori teratas: ${hasilBatch.topCategories.join(', ')}`);
|
|
150
|
+
console.log(` - Daerah teratas: ${hasilBatch.topRegions.join(', ')}`);
|
|
151
|
+
console.log(' - Kata kotor terbanyak:');
|
|
152
|
+
hasilBatch.mostFrequentWords.forEach((word) => {
|
|
153
|
+
console.log(` * "${word.word}": ${word.count} kali`);
|
|
154
|
+
});
|
|
155
|
+
|
|
156
|
+
// ===== 10. Filter berdasarkan kategori dan daerah =====
|
|
157
|
+
console.log('\n10. Filter berdasarkan kategori dan daerah:');
|
|
158
|
+
|
|
159
|
+
// Filter untuk kata dari daerah Jawa saja
|
|
160
|
+
const filterJawa = new IDProfanityFilter({
|
|
161
|
+
regions: ['jawa'], // Hanya filter kata dari Jawa
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
const contohTeksJawa = 'Dasar jancok, asu, bangsat, bacot!';
|
|
165
|
+
console.log('- Teks asli:', contohTeksJawa);
|
|
166
|
+
console.log(
|
|
167
|
+
'- Filter khusus kata Jawa:',
|
|
168
|
+
filterJawa.filter(contohTeksJawa).filtered,
|
|
169
|
+
);
|
|
170
|
+
|
|
171
|
+
// Filter untuk kategori sexual saja
|
|
172
|
+
const filterSexual = new IDProfanityFilter({
|
|
173
|
+
categories: ['sexual'], // Hanya filter kata kategori seksual
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
const contohTeksSexual =
|
|
177
|
+
'Kata kotor seperti anjing dan goblok tidak disensor, tapi bokep disensor.';
|
|
178
|
+
console.log('- Teks asli:', contohTeksSexual);
|
|
179
|
+
console.log(
|
|
180
|
+
'- Filter khusus kategori sexual:',
|
|
181
|
+
filterSexual.filter(contohTeksSexual).filtered,
|
|
182
|
+
);
|
|
183
|
+
|
|
184
|
+
console.log('\n============ SELESAI ============');
|
|
@@ -1,31 +0,0 @@
|
|
|
1
|
-
// import { sexual } from './sexual';
|
|
2
|
-
// import { insult } from './insult';
|
|
3
|
-
// import { profanity } from './profanity';
|
|
4
|
-
// import { slur } from './slur';
|
|
5
|
-
// import { drugs } from './drugs';
|
|
6
|
-
// import { disgusting } from './disgusting';
|
|
7
|
-
// import { blasphemy } from './blasphemy';
|
|
8
|
-
|
|
9
|
-
// export const categories = {
|
|
10
|
-
// sexual, // Kata-kata berbau seksual
|
|
11
|
-
// insult, // Kata-kata penghinaan
|
|
12
|
-
// profanity, // Umpatan umum
|
|
13
|
-
// slur, // Perkataan merendahkan berdasarkan identitas
|
|
14
|
-
// drugs, // Terkait narkoba
|
|
15
|
-
// disgusting, // Kata-kata menjijikkan
|
|
16
|
-
// blasphemy, // Penistaan agama
|
|
17
|
-
// };
|
|
18
|
-
|
|
19
|
-
// export { sexual, insult, profanity, slur, drugs, disgusting, blasphemy };
|
|
20
|
-
|
|
21
|
-
// export const allCategoryWords = [
|
|
22
|
-
// ...sexual,
|
|
23
|
-
// ...insult,
|
|
24
|
-
// ...profanity,
|
|
25
|
-
// ...slur,
|
|
26
|
-
// ...drugs,
|
|
27
|
-
// ...disgusting,
|
|
28
|
-
// ...blasphemy,
|
|
29
|
-
// ];
|
|
30
|
-
|
|
31
|
-
// export default allCategoryWords;
|