@sideid/id-profanity-filter 1.9.5 → 1.10.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/.eslintrc.js +44 -16
  2. package/.github/workflows/release.yml +62 -0
  3. package/CONTRIBUTING.md +150 -150
  4. package/LICENSE +21 -21
  5. package/README.md +548 -285
  6. package/dist/config/options.d.ts +24 -0
  7. package/dist/constants/categories/blasphemy.d.ts +4 -0
  8. package/dist/constants/categories/disgusting.d.ts +4 -0
  9. package/dist/constants/categories/drugs.d.ts +4 -0
  10. package/dist/constants/categories/profanity.d.ts +4 -0
  11. package/dist/constants/categories/slur.d.ts +4 -0
  12. package/dist/index.d.ts +2 -0
  13. package/dist/index.esm.js +923 -94
  14. package/dist/index.esm.js.map +1 -1
  15. package/dist/index.js +924 -93
  16. package/dist/index.js.map +1 -1
  17. package/dist/types/index.d.ts +2 -0
  18. package/dist/utils/ahoCorasick.d.ts +36 -0
  19. package/dist/utils/similarityUtils.d.ts +35 -0
  20. package/eslint.config.mjs +40 -0
  21. package/examples/advanced.ts +120 -0
  22. package/examples/basic.ts +71 -52
  23. package/examples/custom-list.ts +140 -0
  24. package/jest.config.mjs +10 -10
  25. package/package.json +3 -2
  26. package/prettierrc +6 -6
  27. package/rollup.config.mjs +35 -35
  28. package/src/config/options.ts +2 -0
  29. package/src/constants/categories/blasphemy.ts +25 -0
  30. package/src/constants/categories/disgusting.ts +82 -0
  31. package/src/constants/categories/drugs.ts +72 -0
  32. package/src/constants/categories/profanity.ts +139 -0
  33. package/src/constants/categories/slur.ts +102 -0
  34. package/src/constants/regions/general.ts +111 -2
  35. package/src/constants/regions/jawa.ts +257 -3
  36. package/src/constants/wordList.ts +15 -8
  37. package/src/core/analyzer.ts +28 -13
  38. package/src/core/filter.ts +178 -37
  39. package/src/core/matcher.ts +146 -69
  40. package/src/index.ts +21 -2
  41. package/src/types/index.ts +4 -2
  42. package/src/utils/ahoCorasick.ts +179 -0
  43. package/src/utils/regexUtils.ts +0 -1
  44. package/src/utils/similarityUtils.ts +239 -7
  45. package/tsconfig.json +115 -115
  46. package/.github/workflows/ci.yml +0 -0
  47. package/src/constants/categories/index.ts +0 -31
  48. package/src/constants/regions/index.ts +0 -62
@@ -0,0 +1,179 @@
1
+ /**
2
+ * Implementasi algoritma Aho-Corasick untuk pencocokan string
3
+ * Algoritma ini memungkinkan pencarian banyak pola dalam satu kali pembacaan teks
4
+ */
5
+
6
+ interface AhoCorasickNode {
7
+ children: Map<string, AhoCorasickNode>;
8
+ fail: AhoCorasickNode | null;
9
+ output: Set<string>;
10
+ depth: number;
11
+ char?: string;
12
+ }
13
+
14
+ export class AhoCorasick {
15
+ private root: AhoCorasickNode;
16
+ private built: boolean = false;
17
+
18
+ constructor() {
19
+ this.root = {
20
+ children: new Map(),
21
+ fail: null,
22
+ output: new Set(),
23
+ depth: 0,
24
+ };
25
+ }
26
+
27
+ /**
28
+ * Menambahkan pola ke dalam trie
29
+ * @param pattern Pola yang akan ditambahkan
30
+ */
31
+ addPattern(pattern: string): void {
32
+ if (this.built) {
33
+ throw new Error("Cannot add patterns after the automaton is built");
34
+ }
35
+
36
+ let node = this.root;
37
+ const normalizedPattern = pattern.toLowerCase();
38
+
39
+ for (let i = 0; i < normalizedPattern.length; i++) {
40
+ const char = normalizedPattern[i];
41
+
42
+ if (!node.children.has(char)) {
43
+ node.children.set(char, {
44
+ children: new Map(),
45
+ fail: null,
46
+ output: new Set(),
47
+ depth: node.depth + 1,
48
+ char,
49
+ });
50
+ }
51
+
52
+ node = node.children.get(char)!;
53
+ }
54
+
55
+ node.output.add(normalizedPattern);
56
+ }
57
+
58
+ /**
59
+ * Membangun fungsi failure
60
+ */
61
+ build(): void {
62
+ if (this.built) return;
63
+
64
+ const queue: AhoCorasickNode[] = [];
65
+
66
+ // Set fail pointer for depth 1 nodes to root
67
+ for (const child of this.root.children.values()) {
68
+ child.fail = this.root;
69
+ queue.push(child);
70
+ }
71
+
72
+ // BFS to build failure links
73
+ while (queue.length > 0) {
74
+ const current = queue.shift()!;
75
+
76
+ for (const [char, child] of current.children.entries()) {
77
+ queue.push(child);
78
+
79
+ let failNode = current.fail;
80
+
81
+ // Find the longest proper suffix that is also a prefix
82
+ while (failNode !== null && !failNode.children.has(char)) {
83
+ failNode = failNode.fail;
84
+ }
85
+
86
+ if (failNode === null) {
87
+ child.fail = this.root;
88
+ } else {
89
+ child.fail = failNode.children.get(char)!;
90
+
91
+ // Add outputs from the fail state to this node
92
+ for (const output of child.fail.output) {
93
+ child.output.add(output);
94
+ }
95
+ }
96
+ }
97
+ }
98
+
99
+ this.built = true;
100
+ }
101
+
102
+ /**
103
+ * Mencari semua kemunculan pola dalam teks
104
+ * @param text Teks yang akan dicari
105
+ * @returns Map pola yang ditemukan dengan jumlah kemunculannya
106
+ */
107
+ search(text: string): Map<string, number> {
108
+ if (!this.built) {
109
+ this.build();
110
+ }
111
+
112
+ const matches = new Map<string, number>();
113
+ const normalizedText = text.toLowerCase();
114
+ let node = this.root;
115
+
116
+ for (let i = 0; i < normalizedText.length; i++) {
117
+ const char = normalizedText[i];
118
+
119
+ // Follow failure links until we find a matching transition or reach root
120
+ while (node !== this.root && !node.children.has(char)) {
121
+ node = node.fail!;
122
+ }
123
+
124
+ // Try to follow the transition
125
+ if (node.children.has(char)) {
126
+ node = node.children.get(char)!;
127
+ }
128
+
129
+ // Check for any matches at this node
130
+ for (const match of node.output) {
131
+ matches.set(match, (matches.get(match) || 0) + 1);
132
+ }
133
+ }
134
+
135
+ return matches;
136
+ }
137
+
138
+ /**
139
+ * Mencari semua kemunculan pola dalam teks dan mengembalikan hanya pola unik
140
+ * @param text Teks yang akan dicari
141
+ * @returns Set pola yang ditemukan
142
+ */
143
+ searchUnique(text: string): Set<string> {
144
+ const matches = this.search(text);
145
+ return new Set(matches.keys());
146
+ }
147
+
148
+ /**
149
+ * Mengecek apakah teks mengandung setidaknya satu pola
150
+ * @param text Teks yang akan dicari
151
+ * @returns Boolean apakah pola ditemukan
152
+ */
153
+ containsAny(text: string): boolean {
154
+ if (!this.built) {
155
+ this.build();
156
+ }
157
+
158
+ const normalizedText = text.toLowerCase();
159
+ let node = this.root;
160
+
161
+ for (let i = 0; i < normalizedText.length; i++) {
162
+ const char = normalizedText[i];
163
+
164
+ while (node !== this.root && !node.children.has(char)) {
165
+ node = node.fail!;
166
+ }
167
+
168
+ if (node.children.has(char)) {
169
+ node = node.children.get(char)!;
170
+ }
171
+
172
+ if (node.output.size > 0) {
173
+ return true;
174
+ }
175
+ }
176
+
177
+ return false;
178
+ }
179
+ }
@@ -72,7 +72,6 @@ export function addLeetSpeakVariations(pattern: string): string {
72
72
  z: ["z", "2"],
73
73
  };
74
74
 
75
- // Ganti tiap karakter dengan variasinya dalam grup character class
76
75
  return pattern
77
76
  .split("")
78
77
  .map((char) => {
@@ -91,6 +91,48 @@ export function findMostSimilar(
91
91
  return mostSimilar;
92
92
  }
93
93
 
94
+ /**
95
+ * Mencari string yang paling mirip dari array menggunakan Levenshtein distance
96
+ *
97
+ * @param target String target
98
+ * @param candidates Array string kandidat
99
+ * @param threshold Minimum kesamaan yang diterima (0-1)
100
+ * @param maxDistance Jarak Levenshtein maksimal yang diterima (default: 3)
101
+ * @returns String yang paling mirip atau null jika tidak ada yang di atas threshold
102
+ */
103
+ export function findMostSimilarWithLevenshtein(
104
+ target: string,
105
+ candidates: string[],
106
+ threshold: number = 0.7,
107
+ maxDistance: number = 3,
108
+ ): string | null {
109
+ if (!candidates.length) return null;
110
+
111
+ let maxSimilarity = 0;
112
+ let minDistance = Infinity;
113
+ let mostSimilar: string | null = null;
114
+
115
+ for (const candidate of candidates) {
116
+ if (Math.abs(target.length - candidate.length) > maxDistance) continue;
117
+
118
+ const distance = levenshteinDistance(target, candidate);
119
+ const similarity = stringSimilarity(target, candidate);
120
+
121
+ if (
122
+ (similarity > maxSimilarity && similarity >= threshold) ||
123
+ (similarity >= threshold && distance < minDistance)
124
+ ) {
125
+ maxSimilarity = similarity;
126
+ minDistance = distance;
127
+ mostSimilar = candidate;
128
+
129
+ if (distance <= 1 || similarity > 0.95) break;
130
+ }
131
+ }
132
+
133
+ return mostSimilar;
134
+ }
135
+
94
136
  /**
95
137
  * Cek apakah string mungkin merupakan variasi dari kata kotor
96
138
  * menggunakan kesamaan string
@@ -156,12 +198,22 @@ export function clusterSimilarWords(
156
198
 
157
199
  /**
158
200
  * Cari kata-kata kotor yang mungkin dari teks menggunakan kesamaan string
201
+ * dengan optimasi untuk mengurangi kompleksitas
159
202
  *
160
203
  * @param text Teks yang akan diperiksa
161
204
  * @param profanityWords Daftar kata kotor
162
205
  * @param threshold Batas minimum kesamaan (default: 0.8)
163
206
  * @returns Array kata yang mungkin merupakan kata kotor
164
207
  */
208
+ /**
209
+ * Cari kata-kata kotor yang mungkin dari teks berdasarkan kemiripan string
210
+ * dengan optimasi biar prosesnya nggak terlalu berat
211
+ *
212
+ * @param text Teks yang mau dicek
213
+ * @param profanityWords Daftar kata-kata kotor/kasar
214
+ * @param threshold Batas minimal kemiripan (default: 0.8)
215
+ * @returns Array kata yang kemungkinan kata kotor/kasar
216
+ */
165
217
  export function findPossibleProfanityBySimiliarity(
166
218
  text: string,
167
219
  profanityWords: string[],
@@ -170,26 +222,206 @@ export function findPossibleProfanityBySimiliarity(
170
222
  const result: Array<{ word: string; original: string; similarity: number }> =
171
223
  [];
172
224
 
173
- // Pisahkan teks menjadi kata-kata
225
+ // Optimasi dengan bikin map kata-kata kotor berdasarkan huruf pertama
226
+ const profanityMap = new Map<string, string[]>();
227
+
228
+ // Kelompokkan kata-kata kotor berdasarkan huruf pertama biar pencariannya lebih cepat
229
+ for (const word of profanityWords) {
230
+ if (word.length < 1) continue;
231
+
232
+ const firstChar = word[0].toLowerCase();
233
+ if (!profanityMap.has(firstChar)) {
234
+ profanityMap.set(firstChar, []);
235
+ }
236
+ profanityMap.get(firstChar)!.push(word);
237
+ }
238
+
174
239
  const words = text.toLowerCase().split(/\s+/);
175
240
 
176
241
  for (const word of words) {
177
- // Lewati kata-kata yang terlalu pendek
178
242
  if (word.length < 3) continue;
179
243
 
180
- for (const profanity of profanityWords) {
244
+ // Cuma bandingin dengan kata-kata kotor yang huruf pertamanya sama
245
+ // atau yang perbedaan panjangnya masih masuk akal
246
+ const firstChar = word[0];
247
+ const candidateWords = profanityMap.get(firstChar) || [];
248
+
249
+ // Cek juga huruf-huruf yang berdekatan (buat antisipasi typo di huruf pertama)
250
+ // Ini opsional tapi bikin deteksinya lebih bagus
251
+ const charCode = firstChar.charCodeAt(0);
252
+ const prevChar = String.fromCharCode(charCode - 1);
253
+ const nextChar = String.fromCharCode(charCode + 1);
254
+
255
+ const adjacentCandidates = [
256
+ ...(profanityMap.get(prevChar) || []),
257
+ ...(profanityMap.get(nextChar) || []),
258
+ ];
259
+
260
+ // Gabungin kandidat-kandidatnya, prioritasin yang huruf pertamanya sama persis
261
+ const allCandidates = [...candidateWords, ...adjacentCandidates];
262
+
263
+ // Filter kandidat berdasarkan perbedaan panjang sebelum ngitung kemiripannya
264
+ const lengthFilteredCandidates = allCandidates.filter(
265
+ (candidate) => Math.abs(candidate.length - word.length) <= 2,
266
+ );
267
+
268
+ // Cari yang paling cocok
269
+ let bestMatch: {
270
+ word: string;
271
+ original: string;
272
+ similarity: number;
273
+ } | null = null;
274
+
275
+ for (const profanity of lengthFilteredCandidates) {
181
276
  const similarity = stringSimilarity(word, profanity);
182
277
 
183
- if (similarity >= threshold) {
184
- result.push({
278
+ if (
279
+ similarity >= threshold &&
280
+ (!bestMatch || similarity > bestMatch.similarity)
281
+ ) {
282
+ bestMatch = {
185
283
  word,
186
284
  original: profanity,
187
285
  similarity,
188
- });
189
- break;
286
+ };
190
287
  }
191
288
  }
289
+
290
+ if (bestMatch) {
291
+ result.push(bestMatch);
292
+ }
192
293
  }
193
294
 
194
295
  return result;
195
296
  }
297
+
298
+ /**
299
+ * Cari kata-kata kotor yang mungkin dari teks menggunakan Levenshtein distance
300
+ *
301
+ * @param text Teks yang akan diperiksa
302
+ * @param profanityWords Daftar kata kotor
303
+ * @param threshold Batas minimum kesamaan (default: 0.8)
304
+ * @param maxDistance Jarak Levenshtein maksimal (default: 2)
305
+ * @returns Array kata yang mungkin merupakan kata kotor
306
+ */
307
+ export function findProfanityByLevenshteinDistance(
308
+ text: string,
309
+ profanityWords: string[],
310
+ threshold: number = 0.8,
311
+ maxDistance: number = 2,
312
+ ): Array<{
313
+ word: string;
314
+ original: string;
315
+ similarity: number;
316
+ distance: number;
317
+ }> {
318
+ const result: Array<{
319
+ word: string;
320
+ original: string;
321
+ similarity: number;
322
+ distance: number;
323
+ }> = [];
324
+
325
+ // map kata-kata kotor dikelompokkan sesuai panjangnya
326
+ const profanityByLength = new Map<number, string[]>();
327
+
328
+ // Kelompokkin kata-kata kotor berdasarkan panjangnya biar pencarian lebih cepat
329
+ for (const word of profanityWords) {
330
+ const length = word.length;
331
+ if (!profanityByLength.has(length)) {
332
+ profanityByLength.set(length, []);
333
+ }
334
+ profanityByLength.get(length)!.push(word);
335
+ }
336
+
337
+ const words = text.toLowerCase().split(/\s+/);
338
+
339
+ for (const word of words) {
340
+ if (word.length < 3) continue;
341
+
342
+ let bestMatch: {
343
+ word: string;
344
+ original: string;
345
+ similarity: number;
346
+ distance: number;
347
+ } | null = null;
348
+
349
+ for (
350
+ let len = Math.max(3, word.length - maxDistance);
351
+ len <= word.length + maxDistance;
352
+ len++
353
+ ) {
354
+ const candidates = profanityByLength.get(len) || [];
355
+
356
+ for (const profanity of candidates) {
357
+ if (!isCharacterCountSimilar(word, profanity, maxDistance)) {
358
+ continue;
359
+ }
360
+
361
+ const distance = levenshteinDistance(word, profanity);
362
+
363
+ if (distance <= maxDistance) {
364
+ const similarity =
365
+ 1 - distance / Math.max(word.length, profanity.length);
366
+
367
+ if (
368
+ similarity >= threshold &&
369
+ (!bestMatch || similarity > bestMatch.similarity)
370
+ ) {
371
+ bestMatch = {
372
+ word,
373
+ original: profanity,
374
+ similarity,
375
+ distance,
376
+ };
377
+
378
+ if (distance === 0 || similarity > 0.95) {
379
+ break;
380
+ }
381
+ }
382
+ }
383
+ }
384
+ }
385
+
386
+ if (bestMatch) {
387
+ result.push(bestMatch);
388
+ }
389
+ }
390
+
391
+ return result;
392
+ }
393
+
394
+ /**
395
+ * Helper function to efficiently check if character counts between two strings
396
+ * are similar enough to warrant a full Levenshtein calculation
397
+ */
398
+ function isCharacterCountSimilar(
399
+ str1: string,
400
+ str2: string,
401
+ maxDifference: number,
402
+ ): boolean {
403
+ const charCount1: Record<string, number> = {};
404
+ const charCount2: Record<string, number> = {};
405
+
406
+ for (const char of str1) {
407
+ charCount1[char] = (charCount1[char] || 0) + 1;
408
+ }
409
+
410
+ for (const char of str2) {
411
+ charCount2[char] = (charCount2[char] || 0) + 1;
412
+ }
413
+
414
+ let diffCount = 0;
415
+
416
+ for (const char in charCount1) {
417
+ diffCount += Math.abs((charCount1[char] || 0) - (charCount2[char] || 0));
418
+ }
419
+
420
+ for (const char in charCount2) {
421
+ if (!charCount1[char]) {
422
+ diffCount += charCount2[char];
423
+ }
424
+ }
425
+
426
+ return diffCount <= maxDifference * 2;
427
+ }