@sideid/id-profanity-filter 1.10.6 → 1.11.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/.eslintrc.js +44 -44
  2. package/.github/workflows/release.yml +62 -0
  3. package/CONTRIBUTING.md +150 -150
  4. package/LICENSE +21 -21
  5. package/README.md +548 -548
  6. package/dist/index.d.ts +989 -0
  7. package/dist/index.esm.js +545 -50
  8. package/dist/index.esm.js.map +1 -1
  9. package/dist/index.js +545 -50
  10. package/dist/index.js.map +1 -1
  11. package/dist/types/constants/categories/blasphemy.d.ts +4 -0
  12. package/dist/types/constants/categories/disgusting.d.ts +4 -0
  13. package/dist/types/constants/categories/drugs.d.ts +4 -0
  14. package/dist/types/constants/categories/profanity.d.ts +4 -0
  15. package/dist/types/constants/categories/slur.d.ts +4 -0
  16. package/dist/{constants → types/constants}/wordList.d.ts +1 -1
  17. package/dist/{core → types/core}/analyzer.d.ts +1 -1
  18. package/dist/{core → types/core}/filter.d.ts +1 -1
  19. package/dist/{core → types/core}/matcher.d.ts +1 -1
  20. package/dist/types/index.d.ts +375 -57
  21. package/dist/types/types/index.d.ts +59 -0
  22. package/dist/types/utils/ahoCorasick.d.ts +36 -0
  23. package/dist/{utils → types/utils}/similarityUtils.d.ts +10 -0
  24. package/eslint.config.mjs +40 -40
  25. package/examples/advanced.ts +120 -120
  26. package/examples/basic.ts +71 -71
  27. package/examples/custom-list.ts +140 -140
  28. package/jest.config.mjs +10 -10
  29. package/package.json +2 -1
  30. package/prettierrc +6 -6
  31. package/rollup.config.mjs +40 -35
  32. package/src/constants/categories/blasphemy.ts +25 -0
  33. package/src/constants/categories/disgusting.ts +82 -0
  34. package/src/constants/categories/drugs.ts +72 -0
  35. package/src/constants/categories/profanity.ts +139 -0
  36. package/src/constants/categories/slur.ts +102 -0
  37. package/src/constants/regions/general.ts +9 -0
  38. package/src/constants/regions/jawa.ts +356 -354
  39. package/src/core/analyzer.ts +28 -7
  40. package/src/core/matcher.ts +25 -18
  41. package/src/index.ts +15 -15
  42. package/src/utils/ahoCorasick.ts +179 -0
  43. package/src/utils/similarityUtils.ts +157 -20
  44. package/tsconfig.json +115 -115
  45. package/.github/workflows/ci.yml +0 -0
  46. package/dist/constants/categories/index.d.ts +0 -9
  47. package/dist/constants/regions/index.d.ts +0 -8
  48. package/test.js +0 -184
  49. /package/dist/{config → types/config}/options.d.ts +0 -0
  50. /package/dist/{constants → types/constants}/categories/insult.d.ts +0 -0
  51. /package/dist/{constants → types/constants}/categories/sexual.d.ts +0 -0
  52. /package/dist/{constants → types/constants}/regions/batak.d.ts +0 -0
  53. /package/dist/{constants → types/constants}/regions/betawi.d.ts +0 -0
  54. /package/dist/{constants → types/constants}/regions/general.d.ts +0 -0
  55. /package/dist/{constants → types/constants}/regions/jawa.d.ts +0 -0
  56. /package/dist/{constants → types/constants}/regions/sunda.d.ts +0 -0
  57. /package/dist/{utils → types/utils}/regexUtils.d.ts +0 -0
  58. /package/dist/{utils → types/utils}/stringUtils.d.ts +0 -0
@@ -12,7 +12,10 @@ import {
12
12
  calculateSeverity,
13
13
  } from "./matcher";
14
14
  import { splitIntoSentences } from "../utils/stringUtils";
15
- import { findPossibleProfanityBySimiliarity } from "../utils/similarityUtils";
15
+ import {
16
+ findPossibleProfanityBySimiliarity,
17
+ findProfanityByLevenshteinDistance,
18
+ } from "../utils/similarityUtils";
16
19
  import { createContextRegex } from "../utils/regexUtils";
17
20
  import { DEFAULT_OPTIONS } from "../config/options";
18
21
 
@@ -54,12 +57,30 @@ export function analyze(
54
57
  }> = [];
55
58
 
56
59
  if (mergedOptions.detectSimilarity) {
57
- const wordList = matchDetails.map((word) => word.word);
58
- similarWords = findPossibleProfanityBySimiliarity(
59
- text,
60
- wordList,
61
- mergedOptions.similarityThreshold || 0.8,
62
- );
60
+ if (matchDetails.length > 0) {
61
+ const wordList = matchDetails.map((word) => word.word);
62
+
63
+ if (mergedOptions.useLevenshtein) {
64
+ const levenshteinResults = findProfanityByLevenshteinDistance(
65
+ text,
66
+ wordList,
67
+ mergedOptions.similarityThreshold || 0.8,
68
+ mergedOptions.maxLevenshteinDistance || 2,
69
+ );
70
+
71
+ similarWords = levenshteinResults.map((item) => ({
72
+ word: item.word,
73
+ original: item.original,
74
+ similarity: item.similarity,
75
+ }));
76
+ } else {
77
+ similarWords = findPossibleProfanityBySimiliarity(
78
+ text,
79
+ wordList,
80
+ mergedOptions.similarityThreshold || 0.8,
81
+ );
82
+ }
83
+ }
63
84
  }
64
85
 
65
86
  return {
@@ -13,6 +13,21 @@ import {
13
13
  findProfanityByLevenshteinDistance,
14
14
  } from "../utils/similarityUtils";
15
15
  import { DEFAULT_OPTIONS } from "../config/options";
16
+ import { AhoCorasick } from "../utils/ahoCorasick";
17
+
18
+ const globalAhoCorasick = new AhoCorasick();
19
+ let ahoCorasickInitialized = false;
20
+
21
+ function initializeAhoCorasick(words: string[]) {
22
+ if (ahoCorasickInitialized) return;
23
+
24
+ for (const word of words) {
25
+ globalAhoCorasick.addPattern(word);
26
+ }
27
+
28
+ globalAhoCorasick.build();
29
+ ahoCorasickInitialized = true;
30
+ }
16
31
 
17
32
  interface FindProfanityFunction {
18
33
  (text: string, options?: FilterOptions): string[];
@@ -85,27 +100,19 @@ export function findProfanity(
85
100
  const matches = new Set<string>();
86
101
  const actualMatches = new Map<string, string[]>();
87
102
 
88
- wordsToCheck.forEach((word) => {
89
- const regex = createWordRegex(word, {
90
- wholeWord: !checkSubstring,
91
- caseSensitive: false,
92
- leetSpeak: false,
93
- detectSplit: false,
94
- indonesianVariation: false,
95
- });
103
+ initializeAhoCorasick(wordsToCheck);
96
104
 
97
- let match;
98
- while ((match = regex.exec(normalizedText)) !== null) {
99
- const originalWord =
100
- aliasMap.get(word.toLowerCase()) || word.toLowerCase();
101
- matches.add(originalWord);
105
+ const basicMatches = globalAhoCorasick.searchUnique(normalizedText);
106
+ for (const match of basicMatches) {
107
+ const originalWord =
108
+ aliasMap.get(match.toLowerCase()) || match.toLowerCase();
109
+ matches.add(originalWord);
102
110
 
103
- if (!actualMatches.has(originalWord)) {
104
- actualMatches.set(originalWord, []);
105
- }
106
- actualMatches.get(originalWord)?.push(match[0]);
111
+ if (!actualMatches.has(originalWord)) {
112
+ actualMatches.set(originalWord, []);
107
113
  }
108
- });
114
+ actualMatches.get(originalWord)?.push(match);
115
+ }
109
116
 
110
117
  if (detectLeetSpeak) {
111
118
  wordsToCheck.forEach((word) => {
package/src/index.ts CHANGED
@@ -1,27 +1,27 @@
1
- export * from "./types";
2
- export * from "./core/matcher";
3
- export * from "./core/filter";
4
- export * from "./core/analyzer";
5
- export * from "./utils/stringUtils";
6
- export * from "./utils/regexUtils";
7
- export * from "./utils/similarityUtils";
8
- export * from "./config/options";
9
-
10
- import { filter, isProfane } from "./core/filter";
1
+ export * from './types';
2
+ export * from './core/matcher';
3
+ export * from './core/filter';
4
+ export * from './core/analyzer';
5
+ export * from './utils/stringUtils';
6
+ export * from './utils/regexUtils';
7
+ export * from './utils/similarityUtils';
8
+ export * from './config/options';
9
+
10
+ import { filter, isProfane } from './core/filter';
11
11
  import {
12
12
  analyze,
13
13
  batchAnalyze,
14
14
  analyzeBySentence,
15
15
  analyzeWithContext,
16
- } from "./core/analyzer";
17
- import { FilterOptions, FilterResult, AnalysisResult } from "./types";
16
+ } from './core/analyzer';
17
+ import { FilterOptions, FilterResult, AnalysisResult } from './types';
18
18
  import {
19
19
  DEFAULT_OPTIONS,
20
20
  FILTER_PRESETS,
21
21
  CATEGORY_PRESETS,
22
22
  REGION_PRESETS,
23
23
  getPresetOptions,
24
- } from "./config/options";
24
+ } from './config/options';
25
25
 
26
26
  export class IDProfanityFilter {
27
27
  private options: FilterOptions;
@@ -187,8 +187,6 @@ export class IDProfanityFilter {
187
187
  }
188
188
  }
189
189
 
190
- export default IDProfanityFilter;
191
-
192
190
  export const idFilter = {
193
191
  filter: (text: string, options?: FilterOptions) =>
194
192
  filter(text, { ...DEFAULT_OPTIONS, ...options }),
@@ -205,3 +203,5 @@ export const idFilter = {
205
203
  region: REGION_PRESETS,
206
204
  },
207
205
  };
206
+
207
+ export default IDProfanityFilter;
@@ -0,0 +1,179 @@
1
+ /**
2
+ * Implementasi algoritma Aho-Corasick untuk pencocokan string
3
+ * Algoritma ini memungkinkan pencarian banyak pola dalam satu kali pembacaan teks
4
+ */
5
+
6
+ interface AhoCorasickNode {
7
+ children: Map<string, AhoCorasickNode>;
8
+ fail: AhoCorasickNode | null;
9
+ output: Set<string>;
10
+ depth: number;
11
+ char?: string;
12
+ }
13
+
14
+ export class AhoCorasick {
15
+ private root: AhoCorasickNode;
16
+ private built: boolean = false;
17
+
18
+ constructor() {
19
+ this.root = {
20
+ children: new Map(),
21
+ fail: null,
22
+ output: new Set(),
23
+ depth: 0,
24
+ };
25
+ }
26
+
27
+ /**
28
+ * Menambahkan pola ke dalam trie
29
+ * @param pattern Pola yang akan ditambahkan
30
+ */
31
+ addPattern(pattern: string): void {
32
+ if (this.built) {
33
+ throw new Error("Cannot add patterns after the automaton is built");
34
+ }
35
+
36
+ let node = this.root;
37
+ const normalizedPattern = pattern.toLowerCase();
38
+
39
+ for (let i = 0; i < normalizedPattern.length; i++) {
40
+ const char = normalizedPattern[i];
41
+
42
+ if (!node.children.has(char)) {
43
+ node.children.set(char, {
44
+ children: new Map(),
45
+ fail: null,
46
+ output: new Set(),
47
+ depth: node.depth + 1,
48
+ char,
49
+ });
50
+ }
51
+
52
+ node = node.children.get(char)!;
53
+ }
54
+
55
+ node.output.add(normalizedPattern);
56
+ }
57
+
58
+ /**
59
+ * Membangun fungsi failure
60
+ */
61
+ build(): void {
62
+ if (this.built) return;
63
+
64
+ const queue: AhoCorasickNode[] = [];
65
+
66
+ // Set fail pointer for depth 1 nodes to root
67
+ for (const child of this.root.children.values()) {
68
+ child.fail = this.root;
69
+ queue.push(child);
70
+ }
71
+
72
+ // BFS to build failure links
73
+ while (queue.length > 0) {
74
+ const current = queue.shift()!;
75
+
76
+ for (const [char, child] of current.children.entries()) {
77
+ queue.push(child);
78
+
79
+ let failNode = current.fail;
80
+
81
+ // Find the longest proper suffix that is also a prefix
82
+ while (failNode !== null && !failNode.children.has(char)) {
83
+ failNode = failNode.fail;
84
+ }
85
+
86
+ if (failNode === null) {
87
+ child.fail = this.root;
88
+ } else {
89
+ child.fail = failNode.children.get(char)!;
90
+
91
+ // Add outputs from the fail state to this node
92
+ for (const output of child.fail.output) {
93
+ child.output.add(output);
94
+ }
95
+ }
96
+ }
97
+ }
98
+
99
+ this.built = true;
100
+ }
101
+
102
+ /**
103
+ * Mencari semua kemunculan pola dalam teks
104
+ * @param text Teks yang akan dicari
105
+ * @returns Map pola yang ditemukan dengan jumlah kemunculannya
106
+ */
107
+ search(text: string): Map<string, number> {
108
+ if (!this.built) {
109
+ this.build();
110
+ }
111
+
112
+ const matches = new Map<string, number>();
113
+ const normalizedText = text.toLowerCase();
114
+ let node = this.root;
115
+
116
+ for (let i = 0; i < normalizedText.length; i++) {
117
+ const char = normalizedText[i];
118
+
119
+ // Follow failure links until we find a matching transition or reach root
120
+ while (node !== this.root && !node.children.has(char)) {
121
+ node = node.fail!;
122
+ }
123
+
124
+ // Try to follow the transition
125
+ if (node.children.has(char)) {
126
+ node = node.children.get(char)!;
127
+ }
128
+
129
+ // Check for any matches at this node
130
+ for (const match of node.output) {
131
+ matches.set(match, (matches.get(match) || 0) + 1);
132
+ }
133
+ }
134
+
135
+ return matches;
136
+ }
137
+
138
+ /**
139
+ * Mencari semua kemunculan pola dalam teks dan mengembalikan hanya pola unik
140
+ * @param text Teks yang akan dicari
141
+ * @returns Set pola yang ditemukan
142
+ */
143
+ searchUnique(text: string): Set<string> {
144
+ const matches = this.search(text);
145
+ return new Set(matches.keys());
146
+ }
147
+
148
+ /**
149
+ * Mengecek apakah teks mengandung setidaknya satu pola
150
+ * @param text Teks yang akan dicari
151
+ * @returns Boolean apakah pola ditemukan
152
+ */
153
+ containsAny(text: string): boolean {
154
+ if (!this.built) {
155
+ this.build();
156
+ }
157
+
158
+ const normalizedText = text.toLowerCase();
159
+ let node = this.root;
160
+
161
+ for (let i = 0; i < normalizedText.length; i++) {
162
+ const char = normalizedText[i];
163
+
164
+ while (node !== this.root && !node.children.has(char)) {
165
+ node = node.fail!;
166
+ }
167
+
168
+ if (node.children.has(char)) {
169
+ node = node.children.get(char)!;
170
+ }
171
+
172
+ if (node.output.size > 0) {
173
+ return true;
174
+ }
175
+ }
176
+
177
+ return false;
178
+ }
179
+ }
@@ -198,12 +198,22 @@ export function clusterSimilarWords(
198
198
 
199
199
  /**
200
200
  * Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
201
+ * dengan optimasi untuk mengurangi kompleksitas
201
202
  *
202
203
  * @param text Teks yang akan diperiksa
203
204
  * @param profanityWords Daftar kata kotor
204
205
  * @param threshold Batas minimum kesamaan (default: 0.8)
205
206
  * @returns Array kata yang mungkin merupakan kata kotor
206
207
  */
208
+ /**
209
+ * Cari kata-kata kotor yang mungkin dari teks berdasarkan kemiripan string
210
+ * dengan optimasi biar prosesnya nggak terlalu berat
211
+ *
212
+ * @param text Teks yang mau dicek
213
+ * @param profanityWords Daftar kata-kata kotor/kasar
214
+ * @param threshold Batas minimal kemiripan (default: 0.8)
215
+ * @returns Array kata yang kemungkinan kata kotor/kasar
216
+ */
207
217
  export function findPossibleProfanityBySimiliarity(
208
218
  text: string,
209
219
  profanityWords: string[],
@@ -212,23 +222,74 @@ export function findPossibleProfanityBySimiliarity(
212
222
  const result: Array<{ word: string; original: string; similarity: number }> =
213
223
  [];
214
224
 
225
+ // Optimasi dengan bikin map kata-kata kotor berdasarkan huruf pertama
226
+ const profanityMap = new Map<string, string[]>();
227
+
228
+ // Kelompokkan kata-kata kotor berdasarkan huruf pertama biar pencariannya lebih cepat
229
+ for (const word of profanityWords) {
230
+ if (word.length < 1) continue;
231
+
232
+ const firstChar = word[0].toLowerCase();
233
+ if (!profanityMap.has(firstChar)) {
234
+ profanityMap.set(firstChar, []);
235
+ }
236
+ profanityMap.get(firstChar)!.push(word);
237
+ }
238
+
215
239
  const words = text.toLowerCase().split(/\s+/);
216
240
 
217
241
  for (const word of words) {
218
242
  if (word.length < 3) continue;
219
243
 
220
- for (const profanity of profanityWords) {
244
+ // Cuma bandingin dengan kata-kata kotor yang huruf pertamanya sama
245
+ // atau yang perbedaan panjangnya masih masuk akal
246
+ const firstChar = word[0];
247
+ const candidateWords = profanityMap.get(firstChar) || [];
248
+
249
+ // Cek juga huruf-huruf yang berdekatan (buat antisipasi typo di huruf pertama)
250
+ // Ini opsional tapi bikin deteksinya lebih bagus
251
+ const charCode = firstChar.charCodeAt(0);
252
+ const prevChar = String.fromCharCode(charCode - 1);
253
+ const nextChar = String.fromCharCode(charCode + 1);
254
+
255
+ const adjacentCandidates = [
256
+ ...(profanityMap.get(prevChar) || []),
257
+ ...(profanityMap.get(nextChar) || []),
258
+ ];
259
+
260
+ // Gabungin kandidat-kandidatnya, prioritasin yang huruf pertamanya sama persis
261
+ const allCandidates = [...candidateWords, ...adjacentCandidates];
262
+
263
+ // Filter kandidat berdasarkan perbedaan panjang sebelum ngitung kemiripannya
264
+ const lengthFilteredCandidates = allCandidates.filter(
265
+ (candidate) => Math.abs(candidate.length - word.length) <= 2,
266
+ );
267
+
268
+ // Cari yang paling cocok
269
+ let bestMatch: {
270
+ word: string;
271
+ original: string;
272
+ similarity: number;
273
+ } | null = null;
274
+
275
+ for (const profanity of lengthFilteredCandidates) {
221
276
  const similarity = stringSimilarity(word, profanity);
222
277
 
223
- if (similarity >= threshold) {
224
- result.push({
278
+ if (
279
+ similarity >= threshold &&
280
+ (!bestMatch || similarity > bestMatch.similarity)
281
+ ) {
282
+ bestMatch = {
225
283
  word,
226
284
  original: profanity,
227
285
  similarity,
228
- });
229
- break;
286
+ };
230
287
  }
231
288
  }
289
+
290
+ if (bestMatch) {
291
+ result.push(bestMatch);
292
+ }
232
293
  }
233
294
 
234
295
  return result;
@@ -261,30 +322,106 @@ export function findProfanityByLevenshteinDistance(
261
322
  distance: number;
262
323
  }> = [];
263
324
 
325
+ // map kata-kata kotor dikelompokkan sesuai panjangnya
326
+ const profanityByLength = new Map<number, string[]>();
327
+
328
+ // Kelompokkin kata-kata kotor berdasarkan panjangnya biar pencarian lebih cepat
329
+ for (const word of profanityWords) {
330
+ const length = word.length;
331
+ if (!profanityByLength.has(length)) {
332
+ profanityByLength.set(length, []);
333
+ }
334
+ profanityByLength.get(length)!.push(word);
335
+ }
336
+
264
337
  const words = text.toLowerCase().split(/\s+/);
265
338
 
266
339
  for (const word of words) {
267
340
  if (word.length < 3) continue;
268
341
 
269
- for (const profanity of profanityWords) {
270
- if (Math.abs(word.length - profanity.length) > maxDistance) continue;
271
-
272
- const distance = levenshteinDistance(word, profanity);
273
- if (distance <= maxDistance) {
274
- const similarity = stringSimilarity(word, profanity);
275
-
276
- if (similarity >= threshold) {
277
- result.push({
278
- word,
279
- original: profanity,
280
- similarity,
281
- distance,
282
- });
283
- break;
342
+ let bestMatch: {
343
+ word: string;
344
+ original: string;
345
+ similarity: number;
346
+ distance: number;
347
+ } | null = null;
348
+
349
+ for (
350
+ let len = Math.max(3, word.length - maxDistance);
351
+ len <= word.length + maxDistance;
352
+ len++
353
+ ) {
354
+ const candidates = profanityByLength.get(len) || [];
355
+
356
+ for (const profanity of candidates) {
357
+ if (!isCharacterCountSimilar(word, profanity, maxDistance)) {
358
+ continue;
359
+ }
360
+
361
+ const distance = levenshteinDistance(word, profanity);
362
+
363
+ if (distance <= maxDistance) {
364
+ const similarity =
365
+ 1 - distance / Math.max(word.length, profanity.length);
366
+
367
+ if (
368
+ similarity >= threshold &&
369
+ (!bestMatch || similarity > bestMatch.similarity)
370
+ ) {
371
+ bestMatch = {
372
+ word,
373
+ original: profanity,
374
+ similarity,
375
+ distance,
376
+ };
377
+
378
+ if (distance === 0 || similarity > 0.95) {
379
+ break;
380
+ }
381
+ }
284
382
  }
285
383
  }
286
384
  }
385
+
386
+ if (bestMatch) {
387
+ result.push(bestMatch);
388
+ }
287
389
  }
288
390
 
289
391
  return result;
290
392
  }
393
+
394
+ /**
395
+ * Helper function to efficiently check if character counts between two strings
396
+ * are similar enough to warrant a full Levenshtein calculation
397
+ */
398
+ function isCharacterCountSimilar(
399
+ str1: string,
400
+ str2: string,
401
+ maxDifference: number,
402
+ ): boolean {
403
+ const charCount1: Record<string, number> = {};
404
+ const charCount2: Record<string, number> = {};
405
+
406
+ for (const char of str1) {
407
+ charCount1[char] = (charCount1[char] || 0) + 1;
408
+ }
409
+
410
+ for (const char of str2) {
411
+ charCount2[char] = (charCount2[char] || 0) + 1;
412
+ }
413
+
414
+ let diffCount = 0;
415
+
416
+ for (const char in charCount1) {
417
+ diffCount += Math.abs((charCount1[char] || 0) - (charCount2[char] || 0));
418
+ }
419
+
420
+ for (const char in charCount2) {
421
+ if (!charCount1[char]) {
422
+ diffCount += charCount2[char];
423
+ }
424
+ }
425
+
426
+ return diffCount <= maxDifference * 2;
427
+ }