@sideid/id-profanity-filter 1.10.6 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/constants/categories/blasphemy.d.ts +4 -0
- package/dist/constants/categories/disgusting.d.ts +4 -0
- package/dist/constants/categories/drugs.d.ts +4 -0
- package/dist/constants/categories/profanity.d.ts +4 -0
- package/dist/constants/categories/slur.d.ts +4 -0
- package/dist/constants/wordList.d.ts +1 -1
- package/dist/core/analyzer.d.ts +1 -1
- package/dist/core/filter.d.ts +1 -1
- package/dist/core/matcher.d.ts +1 -1
- package/dist/index.esm.js +536 -50
- package/dist/index.esm.js.map +1 -1
- package/dist/index.js +536 -50
- package/dist/index.js.map +1 -1
- package/dist/utils/ahoCorasick.d.ts +36 -0
- package/dist/utils/similarityUtils.d.ts +10 -0
- package/package.json +1 -1
- package/src/constants/categories/blasphemy.ts +25 -0
- package/src/constants/categories/disgusting.ts +82 -0
- package/src/constants/categories/drugs.ts +72 -0
- package/src/constants/categories/profanity.ts +139 -0
- package/src/constants/categories/slur.ts +102 -0
- package/src/constants/regions/jawa.ts +356 -354
- package/src/core/analyzer.ts +28 -7
- package/src/core/matcher.ts +25 -18
- package/src/utils/ahoCorasick.ts +179 -0
- package/src/utils/similarityUtils.ts +157 -20
- package/test.js +0 -184
package/src/core/analyzer.ts
CHANGED
|
@@ -12,7 +12,10 @@ import {
|
|
|
12
12
|
calculateSeverity,
|
|
13
13
|
} from "./matcher";
|
|
14
14
|
import { splitIntoSentences } from "../utils/stringUtils";
|
|
15
|
-
import {
|
|
15
|
+
import {
|
|
16
|
+
findPossibleProfanityBySimiliarity,
|
|
17
|
+
findProfanityByLevenshteinDistance,
|
|
18
|
+
} from "../utils/similarityUtils";
|
|
16
19
|
import { createContextRegex } from "../utils/regexUtils";
|
|
17
20
|
import { DEFAULT_OPTIONS } from "../config/options";
|
|
18
21
|
|
|
@@ -54,12 +57,30 @@ export function analyze(
|
|
|
54
57
|
}> = [];
|
|
55
58
|
|
|
56
59
|
if (mergedOptions.detectSimilarity) {
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
60
|
+
if (matchDetails.length > 0) {
|
|
61
|
+
const wordList = matchDetails.map((word) => word.word);
|
|
62
|
+
|
|
63
|
+
if (mergedOptions.useLevenshtein) {
|
|
64
|
+
const levenshteinResults = findProfanityByLevenshteinDistance(
|
|
65
|
+
text,
|
|
66
|
+
wordList,
|
|
67
|
+
mergedOptions.similarityThreshold || 0.8,
|
|
68
|
+
mergedOptions.maxLevenshteinDistance || 2,
|
|
69
|
+
);
|
|
70
|
+
|
|
71
|
+
similarWords = levenshteinResults.map((item) => ({
|
|
72
|
+
word: item.word,
|
|
73
|
+
original: item.original,
|
|
74
|
+
similarity: item.similarity,
|
|
75
|
+
}));
|
|
76
|
+
} else {
|
|
77
|
+
similarWords = findPossibleProfanityBySimiliarity(
|
|
78
|
+
text,
|
|
79
|
+
wordList,
|
|
80
|
+
mergedOptions.similarityThreshold || 0.8,
|
|
81
|
+
);
|
|
82
|
+
}
|
|
83
|
+
}
|
|
63
84
|
}
|
|
64
85
|
|
|
65
86
|
return {
|
package/src/core/matcher.ts
CHANGED
|
@@ -13,6 +13,21 @@ import {
|
|
|
13
13
|
findProfanityByLevenshteinDistance,
|
|
14
14
|
} from "../utils/similarityUtils";
|
|
15
15
|
import { DEFAULT_OPTIONS } from "../config/options";
|
|
16
|
+
import { AhoCorasick } from "../utils/ahoCorasick";
|
|
17
|
+
|
|
18
|
+
const globalAhoCorasick = new AhoCorasick();
|
|
19
|
+
let ahoCorasickInitialized = false;
|
|
20
|
+
|
|
21
|
+
function initializeAhoCorasick(words: string[]) {
|
|
22
|
+
if (ahoCorasickInitialized) return;
|
|
23
|
+
|
|
24
|
+
for (const word of words) {
|
|
25
|
+
globalAhoCorasick.addPattern(word);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
globalAhoCorasick.build();
|
|
29
|
+
ahoCorasickInitialized = true;
|
|
30
|
+
}
|
|
16
31
|
|
|
17
32
|
interface FindProfanityFunction {
|
|
18
33
|
(text: string, options?: FilterOptions): string[];
|
|
@@ -85,27 +100,19 @@ export function findProfanity(
|
|
|
85
100
|
const matches = new Set<string>();
|
|
86
101
|
const actualMatches = new Map<string, string[]>();
|
|
87
102
|
|
|
88
|
-
wordsToCheck
|
|
89
|
-
const regex = createWordRegex(word, {
|
|
90
|
-
wholeWord: !checkSubstring,
|
|
91
|
-
caseSensitive: false,
|
|
92
|
-
leetSpeak: false,
|
|
93
|
-
detectSplit: false,
|
|
94
|
-
indonesianVariation: false,
|
|
95
|
-
});
|
|
103
|
+
initializeAhoCorasick(wordsToCheck);
|
|
96
104
|
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
105
|
+
const basicMatches = globalAhoCorasick.searchUnique(normalizedText);
|
|
106
|
+
for (const match of basicMatches) {
|
|
107
|
+
const originalWord =
|
|
108
|
+
aliasMap.get(match.toLowerCase()) || match.toLowerCase();
|
|
109
|
+
matches.add(originalWord);
|
|
102
110
|
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
}
|
|
106
|
-
actualMatches.get(originalWord)?.push(match[0]);
|
|
111
|
+
if (!actualMatches.has(originalWord)) {
|
|
112
|
+
actualMatches.set(originalWord, []);
|
|
107
113
|
}
|
|
108
|
-
|
|
114
|
+
actualMatches.get(originalWord)?.push(match);
|
|
115
|
+
}
|
|
109
116
|
|
|
110
117
|
if (detectLeetSpeak) {
|
|
111
118
|
wordsToCheck.forEach((word) => {
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Implementasi algoritma Aho-Corasick untuk pencocokan string
|
|
3
|
+
* Algoritma ini memungkinkan pencarian banyak pola dalam satu kali pembacaan teks
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
interface AhoCorasickNode {
|
|
7
|
+
children: Map<string, AhoCorasickNode>;
|
|
8
|
+
fail: AhoCorasickNode | null;
|
|
9
|
+
output: Set<string>;
|
|
10
|
+
depth: number;
|
|
11
|
+
char?: string;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export class AhoCorasick {
|
|
15
|
+
private root: AhoCorasickNode;
|
|
16
|
+
private built: boolean = false;
|
|
17
|
+
|
|
18
|
+
constructor() {
|
|
19
|
+
this.root = {
|
|
20
|
+
children: new Map(),
|
|
21
|
+
fail: null,
|
|
22
|
+
output: new Set(),
|
|
23
|
+
depth: 0,
|
|
24
|
+
};
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Menambahkan pola ke dalam trie
|
|
29
|
+
* @param pattern Pola yang akan ditambahkan
|
|
30
|
+
*/
|
|
31
|
+
addPattern(pattern: string): void {
|
|
32
|
+
if (this.built) {
|
|
33
|
+
throw new Error("Cannot add patterns after the automaton is built");
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
let node = this.root;
|
|
37
|
+
const normalizedPattern = pattern.toLowerCase();
|
|
38
|
+
|
|
39
|
+
for (let i = 0; i < normalizedPattern.length; i++) {
|
|
40
|
+
const char = normalizedPattern[i];
|
|
41
|
+
|
|
42
|
+
if (!node.children.has(char)) {
|
|
43
|
+
node.children.set(char, {
|
|
44
|
+
children: new Map(),
|
|
45
|
+
fail: null,
|
|
46
|
+
output: new Set(),
|
|
47
|
+
depth: node.depth + 1,
|
|
48
|
+
char,
|
|
49
|
+
});
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
node = node.children.get(char)!;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
node.output.add(normalizedPattern);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Membangun fungsi failure
|
|
60
|
+
*/
|
|
61
|
+
build(): void {
|
|
62
|
+
if (this.built) return;
|
|
63
|
+
|
|
64
|
+
const queue: AhoCorasickNode[] = [];
|
|
65
|
+
|
|
66
|
+
// Set fail pointer for depth 1 nodes to root
|
|
67
|
+
for (const child of this.root.children.values()) {
|
|
68
|
+
child.fail = this.root;
|
|
69
|
+
queue.push(child);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
// BFS to build failure links
|
|
73
|
+
while (queue.length > 0) {
|
|
74
|
+
const current = queue.shift()!;
|
|
75
|
+
|
|
76
|
+
for (const [char, child] of current.children.entries()) {
|
|
77
|
+
queue.push(child);
|
|
78
|
+
|
|
79
|
+
let failNode = current.fail;
|
|
80
|
+
|
|
81
|
+
// Find the longest proper suffix that is also a prefix
|
|
82
|
+
while (failNode !== null && !failNode.children.has(char)) {
|
|
83
|
+
failNode = failNode.fail;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
if (failNode === null) {
|
|
87
|
+
child.fail = this.root;
|
|
88
|
+
} else {
|
|
89
|
+
child.fail = failNode.children.get(char)!;
|
|
90
|
+
|
|
91
|
+
// Add outputs from the fail state to this node
|
|
92
|
+
for (const output of child.fail.output) {
|
|
93
|
+
child.output.add(output);
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
this.built = true;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Mencari semua kemunculan pola dalam teks
|
|
104
|
+
* @param text Teks yang akan dicari
|
|
105
|
+
* @returns Map pola yang ditemukan dengan jumlah kemunculannya
|
|
106
|
+
*/
|
|
107
|
+
search(text: string): Map<string, number> {
|
|
108
|
+
if (!this.built) {
|
|
109
|
+
this.build();
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
const matches = new Map<string, number>();
|
|
113
|
+
const normalizedText = text.toLowerCase();
|
|
114
|
+
let node = this.root;
|
|
115
|
+
|
|
116
|
+
for (let i = 0; i < normalizedText.length; i++) {
|
|
117
|
+
const char = normalizedText[i];
|
|
118
|
+
|
|
119
|
+
// Follow failure links until we find a matching transition or reach root
|
|
120
|
+
while (node !== this.root && !node.children.has(char)) {
|
|
121
|
+
node = node.fail!;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
// Try to follow the transition
|
|
125
|
+
if (node.children.has(char)) {
|
|
126
|
+
node = node.children.get(char)!;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// Check for any matches at this node
|
|
130
|
+
for (const match of node.output) {
|
|
131
|
+
matches.set(match, (matches.get(match) || 0) + 1);
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
return matches;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* Mencari semua kemunculan pola dalam teks dan mengembalikan hanya pola unik
|
|
140
|
+
* @param text Teks yang akan dicari
|
|
141
|
+
* @returns Set pola yang ditemukan
|
|
142
|
+
*/
|
|
143
|
+
searchUnique(text: string): Set<string> {
|
|
144
|
+
const matches = this.search(text);
|
|
145
|
+
return new Set(matches.keys());
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Mengecek apakah teks mengandung setidaknya satu pola
|
|
150
|
+
* @param text Teks yang akan dicari
|
|
151
|
+
* @returns Boolean apakah pola ditemukan
|
|
152
|
+
*/
|
|
153
|
+
containsAny(text: string): boolean {
|
|
154
|
+
if (!this.built) {
|
|
155
|
+
this.build();
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
const normalizedText = text.toLowerCase();
|
|
159
|
+
let node = this.root;
|
|
160
|
+
|
|
161
|
+
for (let i = 0; i < normalizedText.length; i++) {
|
|
162
|
+
const char = normalizedText[i];
|
|
163
|
+
|
|
164
|
+
while (node !== this.root && !node.children.has(char)) {
|
|
165
|
+
node = node.fail!;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
if (node.children.has(char)) {
|
|
169
|
+
node = node.children.get(char)!;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
if (node.output.size > 0) {
|
|
173
|
+
return true;
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
return false;
|
|
178
|
+
}
|
|
179
|
+
}
|
|
@@ -198,12 +198,22 @@ export function clusterSimilarWords(
|
|
|
198
198
|
|
|
199
199
|
/**
|
|
200
200
|
* Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
|
|
201
|
+
* dengan optimasi untuk mengurangi kompleksitas
|
|
201
202
|
*
|
|
202
203
|
* @param text Teks yang akan diperiksa
|
|
203
204
|
* @param profanityWords Daftar kata kotor
|
|
204
205
|
* @param threshold Batas minimum kesamaan (default: 0.8)
|
|
205
206
|
* @returns Array kata yang mungkin merupakan kata kotor
|
|
206
207
|
*/
|
|
208
|
+
/**
|
|
209
|
+
* Cari kata-kata kotor yang mungkin dari teks berdasarkan kemiripan string
|
|
210
|
+
* dengan optimasi biar prosesnya nggak terlalu berat
|
|
211
|
+
*
|
|
212
|
+
* @param text Teks yang mau dicek
|
|
213
|
+
* @param profanityWords Daftar kata-kata kotor/kasar
|
|
214
|
+
* @param threshold Batas minimal kemiripan (default: 0.8)
|
|
215
|
+
* @returns Array kata yang kemungkinan kata kotor/kasar
|
|
216
|
+
*/
|
|
207
217
|
export function findPossibleProfanityBySimiliarity(
|
|
208
218
|
text: string,
|
|
209
219
|
profanityWords: string[],
|
|
@@ -212,23 +222,74 @@ export function findPossibleProfanityBySimiliarity(
|
|
|
212
222
|
const result: Array<{ word: string; original: string; similarity: number }> =
|
|
213
223
|
[];
|
|
214
224
|
|
|
225
|
+
// Optimasi dengan bikin map kata-kata kotor berdasarkan huruf pertama
|
|
226
|
+
const profanityMap = new Map<string, string[]>();
|
|
227
|
+
|
|
228
|
+
// Kelompokkan kata-kata kotor berdasarkan huruf pertama biar pencariannya lebih cepat
|
|
229
|
+
for (const word of profanityWords) {
|
|
230
|
+
if (word.length < 1) continue;
|
|
231
|
+
|
|
232
|
+
const firstChar = word[0].toLowerCase();
|
|
233
|
+
if (!profanityMap.has(firstChar)) {
|
|
234
|
+
profanityMap.set(firstChar, []);
|
|
235
|
+
}
|
|
236
|
+
profanityMap.get(firstChar)!.push(word);
|
|
237
|
+
}
|
|
238
|
+
|
|
215
239
|
const words = text.toLowerCase().split(/\s+/);
|
|
216
240
|
|
|
217
241
|
for (const word of words) {
|
|
218
242
|
if (word.length < 3) continue;
|
|
219
243
|
|
|
220
|
-
|
|
244
|
+
// Cuma bandingin dengan kata-kata kotor yang huruf pertamanya sama
|
|
245
|
+
// atau yang perbedaan panjangnya masih masuk akal
|
|
246
|
+
const firstChar = word[0];
|
|
247
|
+
const candidateWords = profanityMap.get(firstChar) || [];
|
|
248
|
+
|
|
249
|
+
// Cek juga huruf-huruf yang berdekatan (buat antisipasi typo di huruf pertama)
|
|
250
|
+
// Ini opsional tapi bikin deteksinya lebih bagus
|
|
251
|
+
const charCode = firstChar.charCodeAt(0);
|
|
252
|
+
const prevChar = String.fromCharCode(charCode - 1);
|
|
253
|
+
const nextChar = String.fromCharCode(charCode + 1);
|
|
254
|
+
|
|
255
|
+
const adjacentCandidates = [
|
|
256
|
+
...(profanityMap.get(prevChar) || []),
|
|
257
|
+
...(profanityMap.get(nextChar) || []),
|
|
258
|
+
];
|
|
259
|
+
|
|
260
|
+
// Gabungin kandidat-kandidatnya, prioritasin yang huruf pertamanya sama persis
|
|
261
|
+
const allCandidates = [...candidateWords, ...adjacentCandidates];
|
|
262
|
+
|
|
263
|
+
// Filter kandidat berdasarkan perbedaan panjang sebelum ngitung kemiripannya
|
|
264
|
+
const lengthFilteredCandidates = allCandidates.filter(
|
|
265
|
+
(candidate) => Math.abs(candidate.length - word.length) <= 2,
|
|
266
|
+
);
|
|
267
|
+
|
|
268
|
+
// Cari yang paling cocok
|
|
269
|
+
let bestMatch: {
|
|
270
|
+
word: string;
|
|
271
|
+
original: string;
|
|
272
|
+
similarity: number;
|
|
273
|
+
} | null = null;
|
|
274
|
+
|
|
275
|
+
for (const profanity of lengthFilteredCandidates) {
|
|
221
276
|
const similarity = stringSimilarity(word, profanity);
|
|
222
277
|
|
|
223
|
-
if (
|
|
224
|
-
|
|
278
|
+
if (
|
|
279
|
+
similarity >= threshold &&
|
|
280
|
+
(!bestMatch || similarity > bestMatch.similarity)
|
|
281
|
+
) {
|
|
282
|
+
bestMatch = {
|
|
225
283
|
word,
|
|
226
284
|
original: profanity,
|
|
227
285
|
similarity,
|
|
228
|
-
}
|
|
229
|
-
break;
|
|
286
|
+
};
|
|
230
287
|
}
|
|
231
288
|
}
|
|
289
|
+
|
|
290
|
+
if (bestMatch) {
|
|
291
|
+
result.push(bestMatch);
|
|
292
|
+
}
|
|
232
293
|
}
|
|
233
294
|
|
|
234
295
|
return result;
|
|
@@ -261,30 +322,106 @@ export function findProfanityByLevenshteinDistance(
|
|
|
261
322
|
distance: number;
|
|
262
323
|
}> = [];
|
|
263
324
|
|
|
325
|
+
// map kata-kata kotor dikelompokkan sesuai panjangnya
|
|
326
|
+
const profanityByLength = new Map<number, string[]>();
|
|
327
|
+
|
|
328
|
+
// Kelompokkin kata-kata kotor berdasarkan panjangnya biar pencarian lebih cepat
|
|
329
|
+
for (const word of profanityWords) {
|
|
330
|
+
const length = word.length;
|
|
331
|
+
if (!profanityByLength.has(length)) {
|
|
332
|
+
profanityByLength.set(length, []);
|
|
333
|
+
}
|
|
334
|
+
profanityByLength.get(length)!.push(word);
|
|
335
|
+
}
|
|
336
|
+
|
|
264
337
|
const words = text.toLowerCase().split(/\s+/);
|
|
265
338
|
|
|
266
339
|
for (const word of words) {
|
|
267
340
|
if (word.length < 3) continue;
|
|
268
341
|
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
342
|
+
let bestMatch: {
|
|
343
|
+
word: string;
|
|
344
|
+
original: string;
|
|
345
|
+
similarity: number;
|
|
346
|
+
distance: number;
|
|
347
|
+
} | null = null;
|
|
348
|
+
|
|
349
|
+
for (
|
|
350
|
+
let len = Math.max(3, word.length - maxDistance);
|
|
351
|
+
len <= word.length + maxDistance;
|
|
352
|
+
len++
|
|
353
|
+
) {
|
|
354
|
+
const candidates = profanityByLength.get(len) || [];
|
|
355
|
+
|
|
356
|
+
for (const profanity of candidates) {
|
|
357
|
+
if (!isCharacterCountSimilar(word, profanity, maxDistance)) {
|
|
358
|
+
continue;
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
const distance = levenshteinDistance(word, profanity);
|
|
362
|
+
|
|
363
|
+
if (distance <= maxDistance) {
|
|
364
|
+
const similarity =
|
|
365
|
+
1 - distance / Math.max(word.length, profanity.length);
|
|
366
|
+
|
|
367
|
+
if (
|
|
368
|
+
similarity >= threshold &&
|
|
369
|
+
(!bestMatch || similarity > bestMatch.similarity)
|
|
370
|
+
) {
|
|
371
|
+
bestMatch = {
|
|
372
|
+
word,
|
|
373
|
+
original: profanity,
|
|
374
|
+
similarity,
|
|
375
|
+
distance,
|
|
376
|
+
};
|
|
377
|
+
|
|
378
|
+
if (distance === 0 || similarity > 0.95) {
|
|
379
|
+
break;
|
|
380
|
+
}
|
|
381
|
+
}
|
|
284
382
|
}
|
|
285
383
|
}
|
|
286
384
|
}
|
|
385
|
+
|
|
386
|
+
if (bestMatch) {
|
|
387
|
+
result.push(bestMatch);
|
|
388
|
+
}
|
|
287
389
|
}
|
|
288
390
|
|
|
289
391
|
return result;
|
|
290
392
|
}
|
|
393
|
+
|
|
394
|
+
/**
|
|
395
|
+
* Helper function to efficiently check if character counts between two strings
|
|
396
|
+
* are similar enough to warrant a full Levenshtein calculation
|
|
397
|
+
*/
|
|
398
|
+
function isCharacterCountSimilar(
|
|
399
|
+
str1: string,
|
|
400
|
+
str2: string,
|
|
401
|
+
maxDifference: number,
|
|
402
|
+
): boolean {
|
|
403
|
+
const charCount1: Record<string, number> = {};
|
|
404
|
+
const charCount2: Record<string, number> = {};
|
|
405
|
+
|
|
406
|
+
for (const char of str1) {
|
|
407
|
+
charCount1[char] = (charCount1[char] || 0) + 1;
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
for (const char of str2) {
|
|
411
|
+
charCount2[char] = (charCount2[char] || 0) + 1;
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
let diffCount = 0;
|
|
415
|
+
|
|
416
|
+
for (const char in charCount1) {
|
|
417
|
+
diffCount += Math.abs((charCount1[char] || 0) - (charCount2[char] || 0));
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
for (const char in charCount2) {
|
|
421
|
+
if (!charCount1[char]) {
|
|
422
|
+
diffCount += charCount2[char];
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
|
|
426
|
+
return diffCount <= maxDifference * 2;
|
|
427
|
+
}
|
package/test.js
DELETED
|
@@ -1,184 +0,0 @@
|
|
|
1
|
-
// full-example.js
|
|
2
|
-
const { IDProfanityFilter, idFilter } = require('./dist');
|
|
3
|
-
|
|
4
|
-
/**
|
|
5
|
-
* Contoh Penggunaan Lengkap ID-Profanity-Filter
|
|
6
|
-
* ============================================
|
|
7
|
-
* File ini mencakup semua contoh penggunaan utama library.
|
|
8
|
-
*/
|
|
9
|
-
|
|
10
|
-
console.log('============ CONTOH PENGGUNAAN DASAR ============\n');
|
|
11
|
-
|
|
12
|
-
const filter = new IDProfanityFilter();
|
|
13
|
-
|
|
14
|
-
const teks =
|
|
15
|
-
'Dasar kontol kontol kntl babi asu ngentod kamu, jangan banyak bacot! perek';
|
|
16
|
-
|
|
17
|
-
// ===== 1. Cek apakah teks mengandung kata kotor =====
|
|
18
|
-
const hasProfanity = filter.isProfane(teks);
|
|
19
|
-
console.log('1. Apakah teks mengandung kata kotor?', hasProfanity);
|
|
20
|
-
|
|
21
|
-
// ===== 2. Filter kata kotor (mengganti dengan sensor) =====
|
|
22
|
-
const hasil = filter.filter(teks);
|
|
23
|
-
console.log('\n2. Hasil filter:');
|
|
24
|
-
console.log('- Teks asli:', teks);
|
|
25
|
-
console.log('- Teks tersensor:', hasil.filtered);
|
|
26
|
-
console.log('- Jumlah kata yang disensor:', hasil.censored);
|
|
27
|
-
console.log('- Detail penggantian:', hasil.replacements);
|
|
28
|
-
|
|
29
|
-
// ===== 3. Analisis konten =====
|
|
30
|
-
const analisis = filter.analyze(teks);
|
|
31
|
-
console.log('\n3. Hasil analisis:');
|
|
32
|
-
console.log('- Mengandung kata kotor:', analisis.hasProfanity);
|
|
33
|
-
console.log('- Kata kotor yang ditemukan:', analisis.matches);
|
|
34
|
-
console.log('- Kategori kata kotor:', analisis.categories);
|
|
35
|
-
console.log('- Daerah asal kata kotor:', analisis.regions);
|
|
36
|
-
console.log('- Skor keparahan:', analisis.severityScore.toFixed(2));
|
|
37
|
-
|
|
38
|
-
// console.log('\n============ PENGGUNAAN PRESET ============\n');
|
|
39
|
-
|
|
40
|
-
// // ===== 4. Menggunakan preset filter =====
|
|
41
|
-
// console.log('4. Menggunakan preset filter:');
|
|
42
|
-
|
|
43
|
-
// // Preset strict
|
|
44
|
-
// filter.usePreset('strict');
|
|
45
|
-
// console.log('- Preset strict:', filter.filter(teks).filtered);
|
|
46
|
-
|
|
47
|
-
// // Preset childSafe
|
|
48
|
-
// filter.usePreset('childSafe');
|
|
49
|
-
// console.log('- Preset childSafe:', filter.filter(teks).filtered);
|
|
50
|
-
|
|
51
|
-
// // Preset light
|
|
52
|
-
// filter.usePreset('light');
|
|
53
|
-
// console.log('- Preset light:', filter.filter(teks).filtered);
|
|
54
|
-
|
|
55
|
-
console.log('\n============ KUSTOMISASI FILTER ============\n');
|
|
56
|
-
|
|
57
|
-
// ===== 5. Kustomisasi filter =====
|
|
58
|
-
console.log('5. Kustomisasi filter:');
|
|
59
|
-
|
|
60
|
-
filter.setOptions({
|
|
61
|
-
replaceWith: '#', // Menggunakan # sebagai karakter pengganti
|
|
62
|
-
fullWordCensor: false, // Hanya menyensor sebagian kata
|
|
63
|
-
keepFirstAndLast: true, // Menyimpan huruf pertama dan terakhir
|
|
64
|
-
});
|
|
65
|
-
|
|
66
|
-
console.log('- Filter dengan opsi kustom:', filter.filter(teks).filtered);
|
|
67
|
-
|
|
68
|
-
// Mendeteksi variasi penulisan
|
|
69
|
-
filter.setOptions({
|
|
70
|
-
detectLeetSpeak: true,
|
|
71
|
-
indonesianVariation: true,
|
|
72
|
-
detectSplit: true,
|
|
73
|
-
detectSimilarity: true,
|
|
74
|
-
useLevenshtein: true,
|
|
75
|
-
similarityThreshold: 0.85,
|
|
76
|
-
maxLevenshteinDistance: 2,
|
|
77
|
-
});
|
|
78
|
-
|
|
79
|
-
const teks2 =
|
|
80
|
-
'Dasar b4b1 kamu, j4nc0k! Ngent0d! k-o-n-t-o-l! k o n t o l k-0nt0l! anjiing kontool kwontol';
|
|
81
|
-
console.log('- Teks asli dengan variasi:', teks2);
|
|
82
|
-
console.log('- Hasil filter variasi:', filter.filter(teks2).filtered);
|
|
83
|
-
|
|
84
|
-
// ===== 6. Menggunakan whitelist =====
|
|
85
|
-
console.log('\n6. Menggunakan whitelist:');
|
|
86
|
-
|
|
87
|
-
filter.addToWhitelist('anjing');
|
|
88
|
-
const tekstBinatang =
|
|
89
|
-
'Anjing itu hewan peliharaan yang setia, tidak seperti bajingan itu';
|
|
90
|
-
|
|
91
|
-
console.log('- Teks dengan "anjing" dalam konteks binatang:', tekstBinatang);
|
|
92
|
-
console.log(
|
|
93
|
-
'- Hasil filter dengan whitelist:',
|
|
94
|
-
filter.filter(tekstBinatang).filtered,
|
|
95
|
-
);
|
|
96
|
-
|
|
97
|
-
// ===== 7. Menggunakan daftar kata kustom =====
|
|
98
|
-
console.log('\n7. Menggunakan daftar kata kustom:');
|
|
99
|
-
|
|
100
|
-
const customBadWords = ['jelek', 'buruk', 'sampah', 'payah'];
|
|
101
|
-
|
|
102
|
-
// Reset filter dan gunakan daftar kustom
|
|
103
|
-
filter.setOptions({
|
|
104
|
-
wordList: customBadWords,
|
|
105
|
-
replaceWith: '*',
|
|
106
|
-
});
|
|
107
|
-
|
|
108
|
-
const teksKustom = 'Film ini jelek dan payah sekali!';
|
|
109
|
-
console.log('- Teks asli:', teksKustom);
|
|
110
|
-
console.log('- Hasil filter kustom:', filter.filter(teksKustom).filtered);
|
|
111
|
-
|
|
112
|
-
console.log('\n============ ANALISIS LANJUTAN ============\n');
|
|
113
|
-
|
|
114
|
-
// ===== 8. Analisis per kalimat =====
|
|
115
|
-
console.log('8. Analisis per kalimat:');
|
|
116
|
-
|
|
117
|
-
filter.setOptions({}); // Reset ke default
|
|
118
|
-
const kalimat =
|
|
119
|
-
'Saya sangat suka filmnya. Tapi pemainnya seperti anjing, aktingnya buruk.';
|
|
120
|
-
console.log('- Kalimat:', kalimat);
|
|
121
|
-
|
|
122
|
-
const kalimatAnalisis = filter.analyzeBySentence(kalimat);
|
|
123
|
-
console.log('- Hasil analisis per kalimat:');
|
|
124
|
-
kalimatAnalisis.forEach((hasil, index) => {
|
|
125
|
-
console.log(` Kalimat ${index + 1}: "${hasil.sentence}"`);
|
|
126
|
-
console.log(` Mengandung kata kotor: ${hasil.hasProfanity}`);
|
|
127
|
-
if (hasil.hasProfanity) {
|
|
128
|
-
console.log(` Kata kotor: ${hasil.matches.join(', ')}`);
|
|
129
|
-
}
|
|
130
|
-
});
|
|
131
|
-
|
|
132
|
-
// ===== 9. Analisis batch untuk komentar =====
|
|
133
|
-
console.log('\n9. Analisis batch:');
|
|
134
|
-
|
|
135
|
-
const komentar = [
|
|
136
|
-
'Film ini sangat bagus, ceritanya menarik sekali!',
|
|
137
|
-
'Dasar goblok, sialan kamu!',
|
|
138
|
-
'Anjing emang filmnya, sampah banget.',
|
|
139
|
-
];
|
|
140
|
-
|
|
141
|
-
console.log('- Batch teks:');
|
|
142
|
-
komentar.forEach((k, i) => console.log(` ${i + 1}. "${k}"`));
|
|
143
|
-
|
|
144
|
-
const hasilBatch = filter.batchAnalyze(komentar);
|
|
145
|
-
console.log('- Hasil analisis batch:');
|
|
146
|
-
console.log(` - Total teks: ${hasilBatch.totalTexts}`);
|
|
147
|
-
console.log(` - Teks mengandung kata kotor: ${hasilBatch.profaneTexts}`);
|
|
148
|
-
console.log(` - Teks bersih: ${hasilBatch.cleanTexts}`);
|
|
149
|
-
console.log(` - Kategori teratas: ${hasilBatch.topCategories.join(', ')}`);
|
|
150
|
-
console.log(` - Daerah teratas: ${hasilBatch.topRegions.join(', ')}`);
|
|
151
|
-
console.log(' - Kata kotor terbanyak:');
|
|
152
|
-
hasilBatch.mostFrequentWords.forEach((word) => {
|
|
153
|
-
console.log(` * "${word.word}": ${word.count} kali`);
|
|
154
|
-
});
|
|
155
|
-
|
|
156
|
-
// ===== 10. Filter berdasarkan kategori dan daerah =====
|
|
157
|
-
console.log('\n10. Filter berdasarkan kategori dan daerah:');
|
|
158
|
-
|
|
159
|
-
// Filter untuk kata dari daerah Jawa saja
|
|
160
|
-
const filterJawa = new IDProfanityFilter({
|
|
161
|
-
regions: ['jawa'], // Hanya filter kata dari Jawa
|
|
162
|
-
});
|
|
163
|
-
|
|
164
|
-
const contohTeksJawa = 'Dasar jancok, asu, bangsat, bacot!';
|
|
165
|
-
console.log('- Teks asli:', contohTeksJawa);
|
|
166
|
-
console.log(
|
|
167
|
-
'- Filter khusus kata Jawa:',
|
|
168
|
-
filterJawa.filter(contohTeksJawa).filtered,
|
|
169
|
-
);
|
|
170
|
-
|
|
171
|
-
// Filter untuk kategori sexual saja
|
|
172
|
-
const filterSexual = new IDProfanityFilter({
|
|
173
|
-
categories: ['sexual'], // Hanya filter kata kategori seksual
|
|
174
|
-
});
|
|
175
|
-
|
|
176
|
-
const contohTeksSexual =
|
|
177
|
-
'Kata kotor seperti anjing dan goblok tidak disensor, tapi bokep disensor.';
|
|
178
|
-
console.log('- Teks asli:', contohTeksSexual);
|
|
179
|
-
console.log(
|
|
180
|
-
'- Filter khusus kategori sexual:',
|
|
181
|
-
filterSexual.filter(contohTeksSexual).filtered,
|
|
182
|
-
);
|
|
183
|
-
|
|
184
|
-
console.log('\n============ SELESAI ============');
|