@sideid/id-profanity-filter 1.9.5 → 1.10.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/.eslintrc.js +44 -16
  2. package/.github/workflows/release.yml +62 -0
  3. package/CONTRIBUTING.md +150 -150
  4. package/LICENSE +21 -21
  5. package/README.md +548 -285
  6. package/dist/config/options.d.ts +24 -0
  7. package/dist/constants/categories/blasphemy.d.ts +4 -0
  8. package/dist/constants/categories/disgusting.d.ts +4 -0
  9. package/dist/constants/categories/drugs.d.ts +4 -0
  10. package/dist/constants/categories/profanity.d.ts +4 -0
  11. package/dist/constants/categories/slur.d.ts +4 -0
  12. package/dist/index.d.ts +2 -0
  13. package/dist/index.esm.js +923 -94
  14. package/dist/index.esm.js.map +1 -1
  15. package/dist/index.js +924 -93
  16. package/dist/index.js.map +1 -1
  17. package/dist/types/index.d.ts +2 -0
  18. package/dist/utils/ahoCorasick.d.ts +36 -0
  19. package/dist/utils/similarityUtils.d.ts +35 -0
  20. package/eslint.config.mjs +40 -0
  21. package/examples/advanced.ts +120 -0
  22. package/examples/basic.ts +71 -52
  23. package/examples/custom-list.ts +140 -0
  24. package/jest.config.mjs +10 -10
  25. package/package.json +3 -2
  26. package/prettierrc +6 -6
  27. package/rollup.config.mjs +35 -35
  28. package/src/config/options.ts +2 -0
  29. package/src/constants/categories/blasphemy.ts +25 -0
  30. package/src/constants/categories/disgusting.ts +82 -0
  31. package/src/constants/categories/drugs.ts +72 -0
  32. package/src/constants/categories/profanity.ts +139 -0
  33. package/src/constants/categories/slur.ts +102 -0
  34. package/src/constants/regions/general.ts +111 -2
  35. package/src/constants/regions/jawa.ts +257 -3
  36. package/src/constants/wordList.ts +15 -8
  37. package/src/core/analyzer.ts +28 -13
  38. package/src/core/filter.ts +178 -37
  39. package/src/core/matcher.ts +146 -69
  40. package/src/index.ts +21 -2
  41. package/src/types/index.ts +4 -2
  42. package/src/utils/ahoCorasick.ts +179 -0
  43. package/src/utils/regexUtils.ts +0 -1
  44. package/src/utils/similarityUtils.ts +239 -7
  45. package/tsconfig.json +115 -115
  46. package/.github/workflows/ci.yml +0 -0
  47. package/src/constants/categories/index.ts +0 -31
  48. package/src/constants/regions/index.ts +0 -62
@@ -1,12 +1,13 @@
1
1
  import { FilterOptions, FilterResult, ProfanityWord } from "../types";
2
2
  import { findProfanity, findProfanityWithMetadata } from "./matcher";
3
- import { censorWord, escapeRegExp, normalizeText } from "../utils/stringUtils";
3
+ import { censorWord, escapeRegExp } from "../utils/stringUtils";
4
4
  import { createWordRegex } from "../utils/regexUtils";
5
- import {
6
- DEFAULT_OPTIONS,
7
- makeRandomGrawlixString,
8
- getRandomGrawlix,
9
- } from "../config/options";
5
+ import { DEFAULT_OPTIONS, makeRandomGrawlixString } from "../config/options";
6
+
7
+ interface FindProfanityFunction {
8
+ (text: string, options?: FilterOptions): string[];
9
+ lastActualMatches?: Map<string, string[]>;
10
+ }
10
11
 
11
12
  /**
12
13
  * Menyensor kata kotor dalam teks
@@ -28,6 +29,11 @@ export function filter(
28
29
  useRandomGrawlix = false,
29
30
  keepFirstAndLast = false,
30
31
  indonesianVariation = false,
32
+ detectSplit = false,
33
+ detectSimilarity = false,
34
+ useLevenshtein = false,
35
+ maxLevenshteinDistance = 2,
36
+ similarityThreshold = 0.8,
31
37
  } = { ...DEFAULT_OPTIONS, ...options };
32
38
 
33
39
  const matches = findProfanity(text, {
@@ -36,8 +42,16 @@ export function filter(
36
42
  whitelist,
37
43
  checkSubstring,
38
44
  indonesianVariation,
45
+ detectSplit,
46
+ detectSimilarity,
47
+ useLevenshtein,
48
+ maxLevenshteinDistance,
49
+ similarityThreshold,
39
50
  });
40
51
 
52
+ const actualMatches: Map<string, string[]> =
53
+ (findProfanity as FindProfanityFunction).lastActualMatches || new Map();
54
+
41
55
  const matchDetails = findProfanityWithMetadata(text, options);
42
56
 
43
57
  if (matches.length === 0) {
@@ -66,49 +80,176 @@ export function filter(
66
80
  )),
67
81
  );
68
82
 
69
- const regex = createWordRegex(word, {
70
- wholeWord: true,
71
- caseSensitive: false,
72
- leetSpeak: false,
73
- detectSplit: false,
74
- indonesianVariation: false,
75
- });
83
+ const variants = actualMatches.get(word.toLowerCase()) || [];
84
+ variants.push(word);
85
+
86
+ const uniqueVariants = [...new Set(variants)];
76
87
 
77
- let match;
78
- const textToSearch = filteredText;
88
+ uniqueVariants.forEach((variant) => {
89
+ const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
79
90
 
80
- regex.lastIndex = 0;
91
+ let match;
92
+ while ((match = regex.exec(filteredText)) !== null) {
93
+ const originalWord = match[0];
81
94
 
82
- while ((match = regex.exec(textToSearch)) !== null) {
83
- const originalWord = match[0];
95
+ if (whitelist.includes(originalWord.toLowerCase())) continue;
84
96
 
85
- if (whitelist.includes(originalWord.toLowerCase())) continue;
97
+ let censoredWord;
98
+ if (useRandomGrawlix) {
99
+ censoredWord = makeRandomGrawlixString(originalWord.length);
100
+ } else {
101
+ censoredWord = censorWord(
102
+ originalWord,
103
+ replaceWith,
104
+ !fullWordCensor && keepFirstAndLast,
105
+ );
106
+ }
86
107
 
87
- let censoredWord;
88
- if (useRandomGrawlix) {
89
- censoredWord = makeRandomGrawlixString(originalWord.length);
90
- } else {
91
- censoredWord = censorWord(
92
- originalWord,
93
- replaceWith,
94
- !fullWordCensor && keepFirstAndLast,
108
+ replacements.push({
109
+ original: originalWord,
110
+ censored: censoredWord,
111
+ metadata,
112
+ });
113
+
114
+ filteredText = filteredText.replace(
115
+ new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"),
116
+ censoredWord,
95
117
  );
96
118
  }
119
+ });
97
120
 
98
- replacements.push({
99
- original: originalWord,
100
- censored: censoredWord,
101
- metadata,
102
- });
121
+ if (detectSplit || detectLeetSpeak) {
122
+ if (detectLeetSpeak) {
123
+ const leetRegex = createWordRegex(word, {
124
+ wholeWord: true,
125
+ caseSensitive: false,
126
+ leetSpeak: true,
127
+ detectSplit: false,
128
+ indonesianVariation: false,
129
+ });
103
130
 
104
- const replaceRegex = new RegExp(
105
- `\\b${escapeRegExp(originalWord)}\\b`,
106
- "g",
107
- );
108
- filteredText = filteredText.replace(replaceRegex, censoredWord);
131
+ let match;
132
+ while ((match = leetRegex.exec(filteredText)) !== null) {
133
+ const originalWord = match[0];
134
+
135
+ if (whitelist.includes(originalWord.toLowerCase())) continue;
136
+
137
+ let censoredWord;
138
+ if (useRandomGrawlix) {
139
+ censoredWord = makeRandomGrawlixString(originalWord.length);
140
+ } else {
141
+ censoredWord = censorWord(
142
+ originalWord,
143
+ replaceWith,
144
+ !fullWordCensor && keepFirstAndLast,
145
+ );
146
+ }
147
+
148
+ replacements.push({
149
+ original: originalWord,
150
+ censored: censoredWord,
151
+ metadata,
152
+ });
153
+
154
+ filteredText = filteredText.replace(
155
+ new RegExp(escapeRegExp(originalWord), "g"),
156
+ censoredWord,
157
+ );
158
+ }
159
+ }
160
+
161
+ if (detectSplit) {
162
+ const splitRegex = createWordRegex(word, {
163
+ wholeWord: false,
164
+ caseSensitive: false,
165
+ leetSpeak: false,
166
+ detectSplit: true,
167
+ indonesianVariation: false,
168
+ });
169
+
170
+ let match;
171
+ while ((match = splitRegex.exec(filteredText)) !== null) {
172
+ const originalWord = match[0];
173
+
174
+ if (whitelist.includes(originalWord.toLowerCase())) continue;
175
+
176
+ let censoredWord;
177
+ if (useRandomGrawlix) {
178
+ censoredWord = makeRandomGrawlixString(originalWord.length);
179
+ } else {
180
+ censoredWord = censorWord(
181
+ originalWord,
182
+ replaceWith,
183
+ !fullWordCensor && keepFirstAndLast,
184
+ );
185
+ }
186
+
187
+ replacements.push({
188
+ original: originalWord,
189
+ censored: censoredWord,
190
+ metadata,
191
+ });
192
+
193
+ filteredText = filteredText.replace(
194
+ new RegExp(escapeRegExp(originalWord), "g"),
195
+ censoredWord,
196
+ );
197
+ }
198
+ }
109
199
  }
110
200
  });
111
201
 
202
+ if (detectSimilarity && useLevenshtein) {
203
+ matches.forEach((word) => {
204
+ const metadata = matchDetails.find(
205
+ (m) =>
206
+ m.word.toLowerCase() === word.toLowerCase() ||
207
+ (m.aliases &&
208
+ m.aliases.some(
209
+ (alias) => alias.toLowerCase() === word.toLowerCase(),
210
+ )),
211
+ );
212
+
213
+ const variants = actualMatches.get(word.toLowerCase()) || [];
214
+
215
+ variants.forEach((variant) => {
216
+ const exactVariantRegex = new RegExp(
217
+ `\\b${escapeRegExp(variant)}\\b`,
218
+ "gi",
219
+ );
220
+
221
+ let match;
222
+ while ((match = exactVariantRegex.exec(filteredText)) !== null) {
223
+ const originalWord = match[0];
224
+
225
+ if (whitelist.includes(originalWord.toLowerCase())) continue;
226
+
227
+ let censoredWord;
228
+ if (useRandomGrawlix) {
229
+ censoredWord = makeRandomGrawlixString(originalWord.length);
230
+ } else {
231
+ censoredWord = censorWord(
232
+ originalWord,
233
+ replaceWith,
234
+ !fullWordCensor && keepFirstAndLast,
235
+ );
236
+ }
237
+
238
+ replacements.push({
239
+ original: originalWord,
240
+ censored: censoredWord,
241
+ metadata,
242
+ });
243
+
244
+ filteredText = filteredText.replace(
245
+ new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"),
246
+ censoredWord,
247
+ );
248
+ }
249
+ });
250
+ });
251
+ }
252
+
112
253
  return {
113
254
  filtered: filteredText,
114
255
  censored: replacements.length,
@@ -5,24 +5,34 @@ import {
5
5
  FilterOptions,
6
6
  } from "../types";
7
7
 
8
- import { wordObjects, getWordsByFilter } from "../constants/wordList";
9
- import {
10
- normalizeText,
11
- escapeRegExp,
12
- containsAnyWord,
13
- detectSplitWords,
14
- } from "../utils/stringUtils";
15
- import {
16
- createWordRegex,
17
- addLeetSpeakVariations,
18
- addIndonesianVariations,
19
- addSplitVariations,
20
- } from "../utils/regexUtils";
8
+ import { wordObjects } from "../constants/wordList";
9
+ import { normalizeText } from "../utils/stringUtils";
10
+ import { createWordRegex } from "../utils/regexUtils";
21
11
  import {
22
12
  findPossibleProfanityBySimiliarity,
23
- stringSimilarity,
13
+ findProfanityByLevenshteinDistance,
24
14
  } from "../utils/similarityUtils";
25
15
  import { DEFAULT_OPTIONS } from "../config/options";
16
+ import { AhoCorasick } from "../utils/ahoCorasick";
17
+
18
+ const globalAhoCorasick = new AhoCorasick();
19
+ let ahoCorasickInitialized = false;
20
+
21
+ function initializeAhoCorasick(words: string[]) {
22
+ if (ahoCorasickInitialized) return;
23
+
24
+ for (const word of words) {
25
+ globalAhoCorasick.addPattern(word);
26
+ }
27
+
28
+ globalAhoCorasick.build();
29
+ ahoCorasickInitialized = true;
30
+ }
31
+
32
+ interface FindProfanityFunction {
33
+ (text: string, options?: FilterOptions): string[];
34
+ lastActualMatches?: Map<string, string[]>;
35
+ }
26
36
 
27
37
  export function findProfanity(
28
38
  text: string,
@@ -40,53 +50,69 @@ export function findProfanity(
40
50
  detectSimilarity = false,
41
51
  similarityThreshold = 0.8,
42
52
  detectSplit = false,
53
+ useLevenshtein = false,
54
+ maxLevenshteinDistance = 2,
43
55
  } = { ...DEFAULT_OPTIONS, ...options };
44
56
 
45
57
  const normalizedText = normalizeText(text);
46
58
 
47
- let wordsToCheck: string[] = wordList.length > 0 ? wordList : [];
59
+ let baseWordsToCheck: string[] = wordList.length > 0 ? wordList : [];
48
60
 
49
- if (wordsToCheck.length === 0) {
50
- if (categories || regions || severityThreshold > 0) {
51
- wordsToCheck = wordObjects
52
- .filter((word) => {
53
- const matchCategory = categories
54
- ? categories.includes(word.category)
55
- : true;
56
- const matchRegion = regions ? regions.includes(word.region) : true;
57
- const matchSeverity = word.severity >= severityThreshold;
58
- return matchCategory && matchRegion && matchSeverity;
59
- })
60
- .map((word) => word.word);
61
- } else {
62
- wordsToCheck = wordObjects.map((word) => word.word);
63
- }
61
+ if (baseWordsToCheck.length === 0) {
62
+ const filteredWords = wordObjects.filter((word) => {
63
+ const matchCategory = categories
64
+ ? categories.includes(word.category)
65
+ : true;
66
+ const matchRegion = regions ? regions.includes(word.region) : true;
67
+ const matchSeverity = word.severity >= severityThreshold;
68
+ return matchCategory && matchRegion && matchSeverity;
69
+ });
70
+
71
+ baseWordsToCheck = filteredWords.map((word) => word.word);
64
72
  }
65
73
 
66
- wordsToCheck = wordsToCheck.filter(
67
- (word) => !whitelist.includes(word.toLocaleLowerCase()),
68
- );
74
+ const aliasMap = new Map<string, string>();
75
+ wordObjects.forEach((wordObj) => {
76
+ if (wordObj.aliases && wordObj.aliases.length > 0) {
77
+ const matchCategory = categories
78
+ ? categories.includes(wordObj.category)
79
+ : true;
80
+ const matchRegion = regions ? regions.includes(wordObj.region) : true;
81
+ const matchSeverity = wordObj.severity >= severityThreshold;
82
+
83
+ if (matchCategory && matchRegion && matchSeverity) {
84
+ wordObj.aliases.forEach((alias) => {
85
+ aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
86
+ });
87
+ }
88
+ }
89
+ });
90
+
91
+ const wordsToCheck = [
92
+ ...baseWordsToCheck,
93
+ ...Array.from(aliasMap.keys()),
94
+ ].filter((word) => !whitelist.includes(word.toLowerCase()));
69
95
 
70
96
  if (wordsToCheck.length === 0) {
71
97
  return [];
72
98
  }
73
99
 
74
100
  const matches = new Set<string>();
101
+ const actualMatches = new Map<string, string[]>();
75
102
 
76
- wordsToCheck.forEach((word) => {
77
- const regex = createWordRegex(word, {
78
- wholeWord: !checkSubstring,
79
- caseSensitive: false,
80
- leetSpeak: false,
81
- detectSplit: false,
82
- indonesianVariation: false,
83
- });
103
+ initializeAhoCorasick(wordsToCheck);
84
104
 
85
- let match;
86
- while ((match = regex.exec(normalizedText)) !== null) {
87
- matches.add(word.toLowerCase());
105
+ const basicMatches = globalAhoCorasick.searchUnique(normalizedText);
106
+ for (const match of basicMatches) {
107
+ const originalWord =
108
+ aliasMap.get(match.toLowerCase()) || match.toLowerCase();
109
+ matches.add(originalWord);
110
+
111
+ if (!actualMatches.has(originalWord)) {
112
+ actualMatches.set(originalWord, []);
88
113
  }
89
- });
114
+ actualMatches.get(originalWord)?.push(match);
115
+ }
90
116
 
91
117
  if (detectLeetSpeak) {
92
118
  wordsToCheck.forEach((word) => {
@@ -100,7 +126,14 @@ export function findProfanity(
100
126
 
101
127
  let match;
102
128
  while ((match = leetRegex.exec(text)) !== null) {
103
- matches.add(word.toLowerCase());
129
+ const originalWord =
130
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
131
+ matches.add(originalWord);
132
+
133
+ if (!actualMatches.has(originalWord)) {
134
+ actualMatches.set(originalWord, []);
135
+ }
136
+ actualMatches.get(originalWord)?.push(match[0]);
104
137
  }
105
138
  });
106
139
  }
@@ -117,41 +150,85 @@ export function findProfanity(
117
150
 
118
151
  let match;
119
152
  while ((match = variantRegex.exec(text)) !== null) {
120
- matches.add(word.toLowerCase());
153
+ const originalWord =
154
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
155
+ matches.add(originalWord);
156
+
157
+ if (!actualMatches.has(originalWord)) {
158
+ actualMatches.set(originalWord, []);
159
+ }
160
+ actualMatches.get(originalWord)?.push(match[0]);
121
161
  }
122
162
  });
123
163
  }
124
164
 
125
165
  if (detectSplit) {
126
- if (detectSplitWords(text, wordsToCheck)) {
127
- wordsToCheck.forEach((word) => {
128
- const splitRegex = createWordRegex(word, {
129
- wholeWord: false,
130
- caseSensitive: false,
131
- leetSpeak: false,
132
- detectSplit: true,
133
- indonesianVariation: false,
134
- });
166
+ wordsToCheck.forEach((word) => {
167
+ const splitRegex = createWordRegex(word, {
168
+ wholeWord: false,
169
+ caseSensitive: false,
170
+ leetSpeak: false,
171
+ detectSplit: true,
172
+ indonesianVariation: false,
173
+ });
135
174
 
136
- if (splitRegex.test(text)) {
137
- matches.add(word.toLowerCase());
175
+ let match;
176
+ while ((match = splitRegex.exec(text)) !== null) {
177
+ const originalWord =
178
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
179
+ matches.add(originalWord);
180
+
181
+ if (!actualMatches.has(originalWord)) {
182
+ actualMatches.set(originalWord, []);
138
183
  }
139
- });
140
- }
184
+ actualMatches.get(originalWord)?.push(match[0]);
185
+ }
186
+ });
141
187
  }
142
188
 
143
189
  if (detectSimilarity) {
144
- const possibleProfanity = findPossibleProfanityBySimiliarity(
145
- text,
146
- wordsToCheck,
147
- similarityThreshold,
148
- );
149
-
150
- possibleProfanity.forEach((item) => {
151
- matches.add(item.original.toLowerCase());
152
- });
190
+ if (useLevenshtein) {
191
+ const possibleProfanity = findProfanityByLevenshteinDistance(
192
+ text,
193
+ wordsToCheck,
194
+ similarityThreshold,
195
+ maxLevenshteinDistance,
196
+ );
197
+
198
+ possibleProfanity.forEach((item) => {
199
+ const originalWord =
200
+ aliasMap.get(item.original.toLowerCase()) ||
201
+ item.original.toLowerCase();
202
+ matches.add(originalWord);
203
+
204
+ if (!actualMatches.has(originalWord)) {
205
+ actualMatches.set(originalWord, []);
206
+ }
207
+ actualMatches.get(originalWord)?.push(item.word);
208
+ });
209
+ } else {
210
+ const possibleProfanity = findPossibleProfanityBySimiliarity(
211
+ text,
212
+ wordsToCheck,
213
+ similarityThreshold,
214
+ );
215
+
216
+ possibleProfanity.forEach((item) => {
217
+ matches.add(item.original.toLowerCase());
218
+
219
+ const originalWord =
220
+ aliasMap.get(item.original.toLowerCase()) ||
221
+ item.original.toLowerCase();
222
+ if (!actualMatches.has(originalWord)) {
223
+ actualMatches.set(originalWord, []);
224
+ }
225
+ actualMatches.get(originalWord)?.push(item.word);
226
+ });
227
+ }
153
228
  }
154
229
 
230
+ (findProfanity as FindProfanityFunction).lastActualMatches = actualMatches;
231
+
155
232
  return Array.from(matches);
156
233
  }
157
234
 
package/src/index.ts CHANGED
@@ -21,7 +21,6 @@ import {
21
21
  CATEGORY_PRESETS,
22
22
  REGION_PRESETS,
23
23
  getPresetOptions,
24
- makeRandomGrawlixString,
25
24
  } from "./config/options";
26
25
 
27
26
  export class IDProfanityFilter {
@@ -161,10 +160,30 @@ export class IDProfanityFilter {
161
160
  /**
162
161
  * Mengaktifkan deteksi berdasarkan kesamaan
163
162
  * @param threshold Threshold kesamaan (0-1)
163
+ * @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
164
+ * @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
164
165
  */
165
- enableSimilarityDetection(threshold: number = 0.8) {
166
+ enableSimilarityDetection(
167
+ threshold: number = 0.8,
168
+ useLevenshtein: boolean = false,
169
+ maxLevenshteinDistance: number = 2,
170
+ ) {
171
+ this.options.detectSimilarity = true;
172
+ this.options.similarityThreshold = threshold;
173
+ this.options.useLevenshtein = useLevenshtein;
174
+ this.options.maxLevenshteinDistance = maxLevenshteinDistance;
175
+ }
176
+
177
+ /**
178
+ * Mengaktifkan deteksi berbasis Levenshtein distance
179
+ * @param threshold Threshold kesamaan (0-1)
180
+ * @param maxDistance Jarak maksimal Levenshtein (default: 2)
181
+ */
182
+ enableLevenshteinDetection(threshold: number = 0.8, maxDistance: number = 2) {
166
183
  this.options.detectSimilarity = true;
184
+ this.options.useLevenshtein = true;
167
185
  this.options.similarityThreshold = threshold;
186
+ this.options.maxLevenshteinDistance = maxDistance;
168
187
  }
169
188
  }
170
189
 
@@ -50,9 +50,11 @@ export interface FilterOptions {
50
50
  useRandomGrawlix?: boolean;
51
51
  keepFirstAndLast?: boolean;
52
52
  indonesianVariation?: boolean;
53
- detectSimilarity?: boolean;
54
- similarityThreshold?: number;
53
+ detectSimilarity?: boolean; // Enable/disable Levenshtein distance matching
54
+ similarityThreshold?: number; // Threshold for Levenshtein distance similarity (0-1)
55
55
  detectSplit?: boolean;
56
+ useLevenshtein?: boolean; // New option specifically for Levenshtein algorithm
57
+ maxLevenshteinDistance?: number; // Maximum allowed Levenshtein distance
56
58
  }
57
59
 
58
60
  export interface FilterResult {