@sideid/id-profanity-filter 1.0.0 → 1.9.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -2
- package/src/config/options.ts +178 -0
- package/src/constants/categories/index.ts +31 -31
- package/src/constants/categories/insult.ts +114 -114
- package/src/constants/categories/sexual.ts +99 -100
- package/src/constants/regions/batak.ts +17 -17
- package/src/constants/regions/betawi.ts +39 -39
- package/src/constants/regions/general.ts +111 -111
- package/src/constants/regions/index.ts +62 -62
- package/src/constants/regions/jawa.ts +102 -102
- package/src/constants/regions/sunda.ts +35 -35
- package/src/constants/wordList.ts +125 -125
- package/src/core/analyzer.ts +225 -204
- package/src/core/filter.ts +129 -129
- package/src/core/matcher.ts +259 -267
- package/src/index.ts +188 -133
- package/src/types/index.ts +88 -115
- package/src/utils/regexUtils.ts +203 -0
- package/src/utils/similarityUtils.ts +195 -0
- package/src/utils/stringUtils.ts +213 -0
- package/tsconfig.json +1 -1
- /package/{jest.config.js → jest.config.mjs} +0 -0
- /package/{rollup.config.js → rollup.config.mjs} +0 -0
package/src/core/filter.ts
CHANGED
|
@@ -1,129 +1,129 @@
|
|
|
1
|
-
import { FilterOptions, FilterResult, ProfanityWord } from
|
|
2
|
-
import { findProfanity, findProfanityWithMetadata } from
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
detectLeetSpeak,
|
|
26
|
-
whitelist,
|
|
27
|
-
checkSubstring,
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
*
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
return
|
|
129
|
-
}
|
|
1
|
+
import { FilterOptions, FilterResult, ProfanityWord } from "../types";
|
|
2
|
+
import { findProfanity, findProfanityWithMetadata } from "./matcher";
|
|
3
|
+
import { censorWord, escapeRegExp, normalizeText } from "../utils/stringUtils";
|
|
4
|
+
import { createWordRegex } from "../utils/regexUtils";
|
|
5
|
+
import {
|
|
6
|
+
DEFAULT_OPTIONS,
|
|
7
|
+
makeRandomGrawlixString,
|
|
8
|
+
getRandomGrawlix,
|
|
9
|
+
} from "../config/options";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Menyensor kata kotor dalam teks
|
|
13
|
+
*
|
|
14
|
+
* @param text Teks yang akan disensor
|
|
15
|
+
* @param options Opsi untuk filter
|
|
16
|
+
* @returns FilterResult dengan hasil filter
|
|
17
|
+
*/
|
|
18
|
+
export function filter(
|
|
19
|
+
text: string,
|
|
20
|
+
options: FilterOptions = {},
|
|
21
|
+
): FilterResult {
|
|
22
|
+
const {
|
|
23
|
+
replaceWith = "*",
|
|
24
|
+
fullWordCensor = true,
|
|
25
|
+
detectLeetSpeak = true,
|
|
26
|
+
whitelist = [],
|
|
27
|
+
checkSubstring = false,
|
|
28
|
+
useRandomGrawlix = false,
|
|
29
|
+
keepFirstAndLast = false,
|
|
30
|
+
indonesianVariation = false,
|
|
31
|
+
} = { ...DEFAULT_OPTIONS, ...options };
|
|
32
|
+
|
|
33
|
+
const matches = findProfanity(text, {
|
|
34
|
+
...options,
|
|
35
|
+
detectLeetSpeak,
|
|
36
|
+
whitelist,
|
|
37
|
+
checkSubstring,
|
|
38
|
+
indonesianVariation,
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
const matchDetails = findProfanityWithMetadata(text, options);
|
|
42
|
+
|
|
43
|
+
if (matches.length === 0) {
|
|
44
|
+
return {
|
|
45
|
+
filtered: text,
|
|
46
|
+
censored: 0,
|
|
47
|
+
replacements: [],
|
|
48
|
+
};
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
let filteredText = text;
|
|
52
|
+
|
|
53
|
+
const replacements: Array<{
|
|
54
|
+
original: string;
|
|
55
|
+
censored: string;
|
|
56
|
+
metadata?: ProfanityWord;
|
|
57
|
+
}> = [];
|
|
58
|
+
|
|
59
|
+
matches.forEach((word) => {
|
|
60
|
+
const metadata = matchDetails.find(
|
|
61
|
+
(m) =>
|
|
62
|
+
m.word.toLowerCase() === word.toLowerCase() ||
|
|
63
|
+
(m.aliases &&
|
|
64
|
+
m.aliases.some(
|
|
65
|
+
(alias) => alias.toLowerCase() === word.toLowerCase(),
|
|
66
|
+
)),
|
|
67
|
+
);
|
|
68
|
+
|
|
69
|
+
const regex = createWordRegex(word, {
|
|
70
|
+
wholeWord: true,
|
|
71
|
+
caseSensitive: false,
|
|
72
|
+
leetSpeak: false,
|
|
73
|
+
detectSplit: false,
|
|
74
|
+
indonesianVariation: false,
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
let match;
|
|
78
|
+
const textToSearch = filteredText;
|
|
79
|
+
|
|
80
|
+
regex.lastIndex = 0;
|
|
81
|
+
|
|
82
|
+
while ((match = regex.exec(textToSearch)) !== null) {
|
|
83
|
+
const originalWord = match[0];
|
|
84
|
+
|
|
85
|
+
if (whitelist.includes(originalWord.toLowerCase())) continue;
|
|
86
|
+
|
|
87
|
+
let censoredWord;
|
|
88
|
+
if (useRandomGrawlix) {
|
|
89
|
+
censoredWord = makeRandomGrawlixString(originalWord.length);
|
|
90
|
+
} else {
|
|
91
|
+
censoredWord = censorWord(
|
|
92
|
+
originalWord,
|
|
93
|
+
replaceWith,
|
|
94
|
+
!fullWordCensor && keepFirstAndLast,
|
|
95
|
+
);
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
replacements.push({
|
|
99
|
+
original: originalWord,
|
|
100
|
+
censored: censoredWord,
|
|
101
|
+
metadata,
|
|
102
|
+
});
|
|
103
|
+
|
|
104
|
+
const replaceRegex = new RegExp(
|
|
105
|
+
`\\b${escapeRegExp(originalWord)}\\b`,
|
|
106
|
+
"g",
|
|
107
|
+
);
|
|
108
|
+
filteredText = filteredText.replace(replaceRegex, censoredWord);
|
|
109
|
+
}
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
return {
|
|
113
|
+
filtered: filteredText,
|
|
114
|
+
censored: replacements.length,
|
|
115
|
+
replacements,
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Memeriksa apakah teks mengandung kata kotor
|
|
121
|
+
*
|
|
122
|
+
* @param text Teks yang akan diperiksa
|
|
123
|
+
* @param options Opsi untuk pemeriksaan
|
|
124
|
+
* @returns Boolean apakah teks mengandung kata kotor
|
|
125
|
+
*/
|
|
126
|
+
export function isProfane(text: string, options: FilterOptions = {}): boolean {
|
|
127
|
+
const matches = findProfanity(text, options);
|
|
128
|
+
return matches.length > 0;
|
|
129
|
+
}
|