@sideid/id-profanity-filter 1.9.6 → 1.10.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/.eslintrc.js +44 -16
  2. package/.github/workflows/release.yml +62 -0
  3. package/CONTRIBUTING.md +150 -150
  4. package/LICENSE +21 -21
  5. package/README.md +548 -506
  6. package/dist/config/options.d.ts +24 -0
  7. package/dist/constants/categories/blasphemy.d.ts +4 -0
  8. package/dist/constants/categories/disgusting.d.ts +4 -0
  9. package/dist/constants/categories/drugs.d.ts +4 -0
  10. package/dist/constants/categories/profanity.d.ts +4 -0
  11. package/dist/constants/categories/slur.d.ts +4 -0
  12. package/dist/index.d.ts +2 -0
  13. package/dist/index.esm.js +923 -94
  14. package/dist/index.esm.js.map +1 -1
  15. package/dist/index.js +924 -93
  16. package/dist/index.js.map +1 -1
  17. package/dist/types/index.d.ts +2 -0
  18. package/dist/utils/ahoCorasick.d.ts +36 -0
  19. package/dist/utils/similarityUtils.d.ts +35 -0
  20. package/eslint.config.mjs +40 -0
  21. package/examples/advanced.ts +120 -120
  22. package/examples/basic.ts +71 -71
  23. package/examples/custom-list.ts +140 -140
  24. package/jest.config.mjs +10 -10
  25. package/package.json +3 -2
  26. package/prettierrc +6 -6
  27. package/rollup.config.mjs +35 -35
  28. package/src/config/options.ts +2 -0
  29. package/src/constants/categories/blasphemy.ts +25 -0
  30. package/src/constants/categories/disgusting.ts +82 -0
  31. package/src/constants/categories/drugs.ts +72 -0
  32. package/src/constants/categories/profanity.ts +139 -0
  33. package/src/constants/categories/slur.ts +102 -0
  34. package/src/constants/regions/general.ts +111 -2
  35. package/src/constants/regions/jawa.ts +257 -3
  36. package/src/constants/wordList.ts +15 -8
  37. package/src/core/analyzer.ts +28 -13
  38. package/src/core/filter.ts +178 -37
  39. package/src/core/matcher.ts +146 -69
  40. package/src/index.ts +21 -2
  41. package/src/types/index.ts +4 -2
  42. package/src/utils/ahoCorasick.ts +179 -0
  43. package/src/utils/regexUtils.ts +0 -1
  44. package/src/utils/similarityUtils.ts +239 -7
  45. package/tsconfig.json +115 -115
  46. package/.github/workflows/ci.yml +0 -0
  47. package/dist/constants/categories/index.d.ts +0 -9
  48. package/dist/constants/regions/index.d.ts +0 -8
  49. package/src/constants/categories/index.ts +0 -31
  50. package/src/constants/regions/index.ts +0 -62
@@ -5,24 +5,34 @@ import {
5
5
  FilterOptions,
6
6
  } from "../types";
7
7
 
8
- import { wordObjects, getWordsByFilter } from "../constants/wordList";
9
- import {
10
- normalizeText,
11
- escapeRegExp,
12
- containsAnyWord,
13
- detectSplitWords,
14
- } from "../utils/stringUtils";
15
- import {
16
- createWordRegex,
17
- addLeetSpeakVariations,
18
- addIndonesianVariations,
19
- addSplitVariations,
20
- } from "../utils/regexUtils";
8
+ import { wordObjects } from "../constants/wordList";
9
+ import { normalizeText } from "../utils/stringUtils";
10
+ import { createWordRegex } from "../utils/regexUtils";
21
11
  import {
22
12
  findPossibleProfanityBySimiliarity,
23
- stringSimilarity,
13
+ findProfanityByLevenshteinDistance,
24
14
  } from "../utils/similarityUtils";
25
15
  import { DEFAULT_OPTIONS } from "../config/options";
16
+ import { AhoCorasick } from "../utils/ahoCorasick";
17
+
18
+ const globalAhoCorasick = new AhoCorasick();
19
+ let ahoCorasickInitialized = false;
20
+
21
+ function initializeAhoCorasick(words: string[]) {
22
+ if (ahoCorasickInitialized) return;
23
+
24
+ for (const word of words) {
25
+ globalAhoCorasick.addPattern(word);
26
+ }
27
+
28
+ globalAhoCorasick.build();
29
+ ahoCorasickInitialized = true;
30
+ }
31
+
32
+ interface FindProfanityFunction {
33
+ (text: string, options?: FilterOptions): string[];
34
+ lastActualMatches?: Map<string, string[]>;
35
+ }
26
36
 
27
37
  export function findProfanity(
28
38
  text: string,
@@ -40,53 +50,69 @@ export function findProfanity(
40
50
  detectSimilarity = false,
41
51
  similarityThreshold = 0.8,
42
52
  detectSplit = false,
53
+ useLevenshtein = false,
54
+ maxLevenshteinDistance = 2,
43
55
  } = { ...DEFAULT_OPTIONS, ...options };
44
56
 
45
57
  const normalizedText = normalizeText(text);
46
58
 
47
- let wordsToCheck: string[] = wordList.length > 0 ? wordList : [];
59
+ let baseWordsToCheck: string[] = wordList.length > 0 ? wordList : [];
48
60
 
49
- if (wordsToCheck.length === 0) {
50
- if (categories || regions || severityThreshold > 0) {
51
- wordsToCheck = wordObjects
52
- .filter((word) => {
53
- const matchCategory = categories
54
- ? categories.includes(word.category)
55
- : true;
56
- const matchRegion = regions ? regions.includes(word.region) : true;
57
- const matchSeverity = word.severity >= severityThreshold;
58
- return matchCategory && matchRegion && matchSeverity;
59
- })
60
- .map((word) => word.word);
61
- } else {
62
- wordsToCheck = wordObjects.map((word) => word.word);
63
- }
61
+ if (baseWordsToCheck.length === 0) {
62
+ const filteredWords = wordObjects.filter((word) => {
63
+ const matchCategory = categories
64
+ ? categories.includes(word.category)
65
+ : true;
66
+ const matchRegion = regions ? regions.includes(word.region) : true;
67
+ const matchSeverity = word.severity >= severityThreshold;
68
+ return matchCategory && matchRegion && matchSeverity;
69
+ });
70
+
71
+ baseWordsToCheck = filteredWords.map((word) => word.word);
64
72
  }
65
73
 
66
- wordsToCheck = wordsToCheck.filter(
67
- (word) => !whitelist.includes(word.toLocaleLowerCase()),
68
- );
74
+ const aliasMap = new Map<string, string>();
75
+ wordObjects.forEach((wordObj) => {
76
+ if (wordObj.aliases && wordObj.aliases.length > 0) {
77
+ const matchCategory = categories
78
+ ? categories.includes(wordObj.category)
79
+ : true;
80
+ const matchRegion = regions ? regions.includes(wordObj.region) : true;
81
+ const matchSeverity = wordObj.severity >= severityThreshold;
82
+
83
+ if (matchCategory && matchRegion && matchSeverity) {
84
+ wordObj.aliases.forEach((alias) => {
85
+ aliasMap.set(alias.toLowerCase(), wordObj.word.toLowerCase());
86
+ });
87
+ }
88
+ }
89
+ });
90
+
91
+ const wordsToCheck = [
92
+ ...baseWordsToCheck,
93
+ ...Array.from(aliasMap.keys()),
94
+ ].filter((word) => !whitelist.includes(word.toLowerCase()));
69
95
 
70
96
  if (wordsToCheck.length === 0) {
71
97
  return [];
72
98
  }
73
99
 
74
100
  const matches = new Set<string>();
101
+ const actualMatches = new Map<string, string[]>();
75
102
 
76
- wordsToCheck.forEach((word) => {
77
- const regex = createWordRegex(word, {
78
- wholeWord: !checkSubstring,
79
- caseSensitive: false,
80
- leetSpeak: false,
81
- detectSplit: false,
82
- indonesianVariation: false,
83
- });
103
+ initializeAhoCorasick(wordsToCheck);
84
104
 
85
- let match;
86
- while ((match = regex.exec(normalizedText)) !== null) {
87
- matches.add(word.toLowerCase());
105
+ const basicMatches = globalAhoCorasick.searchUnique(normalizedText);
106
+ for (const match of basicMatches) {
107
+ const originalWord =
108
+ aliasMap.get(match.toLowerCase()) || match.toLowerCase();
109
+ matches.add(originalWord);
110
+
111
+ if (!actualMatches.has(originalWord)) {
112
+ actualMatches.set(originalWord, []);
88
113
  }
89
- });
114
+ actualMatches.get(originalWord)?.push(match);
115
+ }
90
116
 
91
117
  if (detectLeetSpeak) {
92
118
  wordsToCheck.forEach((word) => {
@@ -100,7 +126,14 @@ export function findProfanity(
100
126
 
101
127
  let match;
102
128
  while ((match = leetRegex.exec(text)) !== null) {
103
- matches.add(word.toLowerCase());
129
+ const originalWord =
130
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
131
+ matches.add(originalWord);
132
+
133
+ if (!actualMatches.has(originalWord)) {
134
+ actualMatches.set(originalWord, []);
135
+ }
136
+ actualMatches.get(originalWord)?.push(match[0]);
104
137
  }
105
138
  });
106
139
  }
@@ -117,41 +150,85 @@ export function findProfanity(
117
150
 
118
151
  let match;
119
152
  while ((match = variantRegex.exec(text)) !== null) {
120
- matches.add(word.toLowerCase());
153
+ const originalWord =
154
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
155
+ matches.add(originalWord);
156
+
157
+ if (!actualMatches.has(originalWord)) {
158
+ actualMatches.set(originalWord, []);
159
+ }
160
+ actualMatches.get(originalWord)?.push(match[0]);
121
161
  }
122
162
  });
123
163
  }
124
164
 
125
165
  if (detectSplit) {
126
- if (detectSplitWords(text, wordsToCheck)) {
127
- wordsToCheck.forEach((word) => {
128
- const splitRegex = createWordRegex(word, {
129
- wholeWord: false,
130
- caseSensitive: false,
131
- leetSpeak: false,
132
- detectSplit: true,
133
- indonesianVariation: false,
134
- });
166
+ wordsToCheck.forEach((word) => {
167
+ const splitRegex = createWordRegex(word, {
168
+ wholeWord: false,
169
+ caseSensitive: false,
170
+ leetSpeak: false,
171
+ detectSplit: true,
172
+ indonesianVariation: false,
173
+ });
135
174
 
136
- if (splitRegex.test(text)) {
137
- matches.add(word.toLowerCase());
175
+ let match;
176
+ while ((match = splitRegex.exec(text)) !== null) {
177
+ const originalWord =
178
+ aliasMap.get(word.toLowerCase()) || word.toLowerCase();
179
+ matches.add(originalWord);
180
+
181
+ if (!actualMatches.has(originalWord)) {
182
+ actualMatches.set(originalWord, []);
138
183
  }
139
- });
140
- }
184
+ actualMatches.get(originalWord)?.push(match[0]);
185
+ }
186
+ });
141
187
  }
142
188
 
143
189
  if (detectSimilarity) {
144
- const possibleProfanity = findPossibleProfanityBySimiliarity(
145
- text,
146
- wordsToCheck,
147
- similarityThreshold,
148
- );
149
-
150
- possibleProfanity.forEach((item) => {
151
- matches.add(item.original.toLowerCase());
152
- });
190
+ if (useLevenshtein) {
191
+ const possibleProfanity = findProfanityByLevenshteinDistance(
192
+ text,
193
+ wordsToCheck,
194
+ similarityThreshold,
195
+ maxLevenshteinDistance,
196
+ );
197
+
198
+ possibleProfanity.forEach((item) => {
199
+ const originalWord =
200
+ aliasMap.get(item.original.toLowerCase()) ||
201
+ item.original.toLowerCase();
202
+ matches.add(originalWord);
203
+
204
+ if (!actualMatches.has(originalWord)) {
205
+ actualMatches.set(originalWord, []);
206
+ }
207
+ actualMatches.get(originalWord)?.push(item.word);
208
+ });
209
+ } else {
210
+ const possibleProfanity = findPossibleProfanityBySimiliarity(
211
+ text,
212
+ wordsToCheck,
213
+ similarityThreshold,
214
+ );
215
+
216
+ possibleProfanity.forEach((item) => {
217
+ matches.add(item.original.toLowerCase());
218
+
219
+ const originalWord =
220
+ aliasMap.get(item.original.toLowerCase()) ||
221
+ item.original.toLowerCase();
222
+ if (!actualMatches.has(originalWord)) {
223
+ actualMatches.set(originalWord, []);
224
+ }
225
+ actualMatches.get(originalWord)?.push(item.word);
226
+ });
227
+ }
153
228
  }
154
229
 
230
+ (findProfanity as FindProfanityFunction).lastActualMatches = actualMatches;
231
+
155
232
  return Array.from(matches);
156
233
  }
157
234
 
package/src/index.ts CHANGED
@@ -21,7 +21,6 @@ import {
21
21
  CATEGORY_PRESETS,
22
22
  REGION_PRESETS,
23
23
  getPresetOptions,
24
- makeRandomGrawlixString,
25
24
  } from "./config/options";
26
25
 
27
26
  export class IDProfanityFilter {
@@ -161,10 +160,30 @@ export class IDProfanityFilter {
161
160
  /**
162
161
  * Mengaktifkan deteksi berdasarkan kesamaan
163
162
  * @param threshold Threshold kesamaan (0-1)
163
+ * @param useLevenshtein Gunakan algoritma Levenshtein untuk deteksi
164
+ * @param maxLevenshteinDistance Jarak maksimal Levenshtein (default: 2)
164
165
  */
165
- enableSimilarityDetection(threshold: number = 0.8) {
166
+ enableSimilarityDetection(
167
+ threshold: number = 0.8,
168
+ useLevenshtein: boolean = false,
169
+ maxLevenshteinDistance: number = 2,
170
+ ) {
171
+ this.options.detectSimilarity = true;
172
+ this.options.similarityThreshold = threshold;
173
+ this.options.useLevenshtein = useLevenshtein;
174
+ this.options.maxLevenshteinDistance = maxLevenshteinDistance;
175
+ }
176
+
177
+ /**
178
+ * Mengaktifkan deteksi berbasis Levenshtein distance
179
+ * @param threshold Threshold kesamaan (0-1)
180
+ * @param maxDistance Jarak maksimal Levenshtein (default: 2)
181
+ */
182
+ enableLevenshteinDetection(threshold: number = 0.8, maxDistance: number = 2) {
166
183
  this.options.detectSimilarity = true;
184
+ this.options.useLevenshtein = true;
167
185
  this.options.similarityThreshold = threshold;
186
+ this.options.maxLevenshteinDistance = maxDistance;
168
187
  }
169
188
  }
170
189
 
@@ -50,9 +50,11 @@ export interface FilterOptions {
50
50
  useRandomGrawlix?: boolean;
51
51
  keepFirstAndLast?: boolean;
52
52
  indonesianVariation?: boolean;
53
- detectSimilarity?: boolean;
54
- similarityThreshold?: number;
53
+ detectSimilarity?: boolean; // Enable/disable Levenshtein distance matching
54
+ similarityThreshold?: number; // Threshold for Levenshtein distance similarity (0-1)
55
55
  detectSplit?: boolean;
56
+ useLevenshtein?: boolean; // New option specifically for Levenshtein algorithm
57
+ maxLevenshteinDistance?: number; // Maximum allowed Levenshtein distance
56
58
  }
57
59
 
58
60
  export interface FilterResult {
@@ -0,0 +1,179 @@
1
+ /**
2
+ * Implementasi algoritma Aho-Corasick untuk pencocokan string
3
+ * Algoritma ini memungkinkan pencarian banyak pola dalam satu kali pembacaan teks
4
+ */
5
+
6
+ interface AhoCorasickNode {
7
+ children: Map<string, AhoCorasickNode>;
8
+ fail: AhoCorasickNode | null;
9
+ output: Set<string>;
10
+ depth: number;
11
+ char?: string;
12
+ }
13
+
14
+ export class AhoCorasick {
15
+ private root: AhoCorasickNode;
16
+ private built: boolean = false;
17
+
18
+ constructor() {
19
+ this.root = {
20
+ children: new Map(),
21
+ fail: null,
22
+ output: new Set(),
23
+ depth: 0,
24
+ };
25
+ }
26
+
27
+ /**
28
+ * Menambahkan pola ke dalam trie
29
+ * @param pattern Pola yang akan ditambahkan
30
+ */
31
+ addPattern(pattern: string): void {
32
+ if (this.built) {
33
+ throw new Error("Cannot add patterns after the automaton is built");
34
+ }
35
+
36
+ let node = this.root;
37
+ const normalizedPattern = pattern.toLowerCase();
38
+
39
+ for (let i = 0; i < normalizedPattern.length; i++) {
40
+ const char = normalizedPattern[i];
41
+
42
+ if (!node.children.has(char)) {
43
+ node.children.set(char, {
44
+ children: new Map(),
45
+ fail: null,
46
+ output: new Set(),
47
+ depth: node.depth + 1,
48
+ char,
49
+ });
50
+ }
51
+
52
+ node = node.children.get(char)!;
53
+ }
54
+
55
+ node.output.add(normalizedPattern);
56
+ }
57
+
58
+ /**
59
+ * Membangun fungsi failure
60
+ */
61
+ build(): void {
62
+ if (this.built) return;
63
+
64
+ const queue: AhoCorasickNode[] = [];
65
+
66
+ // Set fail pointer for depth 1 nodes to root
67
+ for (const child of this.root.children.values()) {
68
+ child.fail = this.root;
69
+ queue.push(child);
70
+ }
71
+
72
+ // BFS to build failure links
73
+ while (queue.length > 0) {
74
+ const current = queue.shift()!;
75
+
76
+ for (const [char, child] of current.children.entries()) {
77
+ queue.push(child);
78
+
79
+ let failNode = current.fail;
80
+
81
+ // Find the longest proper suffix that is also a prefix
82
+ while (failNode !== null && !failNode.children.has(char)) {
83
+ failNode = failNode.fail;
84
+ }
85
+
86
+ if (failNode === null) {
87
+ child.fail = this.root;
88
+ } else {
89
+ child.fail = failNode.children.get(char)!;
90
+
91
+ // Add outputs from the fail state to this node
92
+ for (const output of child.fail.output) {
93
+ child.output.add(output);
94
+ }
95
+ }
96
+ }
97
+ }
98
+
99
+ this.built = true;
100
+ }
101
+
102
+ /**
103
+ * Mencari semua kemunculan pola dalam teks
104
+ * @param text Teks yang akan dicari
105
+ * @returns Map pola yang ditemukan dengan jumlah kemunculannya
106
+ */
107
+ search(text: string): Map<string, number> {
108
+ if (!this.built) {
109
+ this.build();
110
+ }
111
+
112
+ const matches = new Map<string, number>();
113
+ const normalizedText = text.toLowerCase();
114
+ let node = this.root;
115
+
116
+ for (let i = 0; i < normalizedText.length; i++) {
117
+ const char = normalizedText[i];
118
+
119
+ // Follow failure links until we find a matching transition or reach root
120
+ while (node !== this.root && !node.children.has(char)) {
121
+ node = node.fail!;
122
+ }
123
+
124
+ // Try to follow the transition
125
+ if (node.children.has(char)) {
126
+ node = node.children.get(char)!;
127
+ }
128
+
129
+ // Check for any matches at this node
130
+ for (const match of node.output) {
131
+ matches.set(match, (matches.get(match) || 0) + 1);
132
+ }
133
+ }
134
+
135
+ return matches;
136
+ }
137
+
138
+ /**
139
+ * Mencari semua kemunculan pola dalam teks dan mengembalikan hanya pola unik
140
+ * @param text Teks yang akan dicari
141
+ * @returns Set pola yang ditemukan
142
+ */
143
+ searchUnique(text: string): Set<string> {
144
+ const matches = this.search(text);
145
+ return new Set(matches.keys());
146
+ }
147
+
148
+ /**
149
+ * Mengecek apakah teks mengandung setidaknya satu pola
150
+ * @param text Teks yang akan dicari
151
+ * @returns Boolean apakah pola ditemukan
152
+ */
153
+ containsAny(text: string): boolean {
154
+ if (!this.built) {
155
+ this.build();
156
+ }
157
+
158
+ const normalizedText = text.toLowerCase();
159
+ let node = this.root;
160
+
161
+ for (let i = 0; i < normalizedText.length; i++) {
162
+ const char = normalizedText[i];
163
+
164
+ while (node !== this.root && !node.children.has(char)) {
165
+ node = node.fail!;
166
+ }
167
+
168
+ if (node.children.has(char)) {
169
+ node = node.children.get(char)!;
170
+ }
171
+
172
+ if (node.output.size > 0) {
173
+ return true;
174
+ }
175
+ }
176
+
177
+ return false;
178
+ }
179
+ }
@@ -72,7 +72,6 @@ export function addLeetSpeakVariations(pattern: string): string {
72
72
  z: ["z", "2"],
73
73
  };
74
74
 
75
- // Ganti tiap karakter dengan variasinya dalam grup character class
76
75
  return pattern
77
76
  .split("")
78
77
  .map((char) => {