@sideid/id-profanity-filter 1.9.6 → 1.10.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/.eslintrc.js +44 -16
  2. package/.github/workflows/release.yml +62 -0
  3. package/CONTRIBUTING.md +150 -150
  4. package/LICENSE +21 -21
  5. package/README.md +548 -506
  6. package/dist/config/options.d.ts +24 -0
  7. package/dist/constants/categories/blasphemy.d.ts +4 -0
  8. package/dist/constants/categories/disgusting.d.ts +4 -0
  9. package/dist/constants/categories/drugs.d.ts +4 -0
  10. package/dist/constants/categories/profanity.d.ts +4 -0
  11. package/dist/constants/categories/slur.d.ts +4 -0
  12. package/dist/index.d.ts +2 -0
  13. package/dist/index.esm.js +923 -94
  14. package/dist/index.esm.js.map +1 -1
  15. package/dist/index.js +924 -93
  16. package/dist/index.js.map +1 -1
  17. package/dist/types/index.d.ts +2 -0
  18. package/dist/utils/ahoCorasick.d.ts +36 -0
  19. package/dist/utils/similarityUtils.d.ts +35 -0
  20. package/eslint.config.mjs +40 -0
  21. package/examples/advanced.ts +120 -120
  22. package/examples/basic.ts +71 -71
  23. package/examples/custom-list.ts +140 -140
  24. package/jest.config.mjs +10 -10
  25. package/package.json +3 -2
  26. package/prettierrc +6 -6
  27. package/rollup.config.mjs +35 -35
  28. package/src/config/options.ts +2 -0
  29. package/src/constants/categories/blasphemy.ts +25 -0
  30. package/src/constants/categories/disgusting.ts +82 -0
  31. package/src/constants/categories/drugs.ts +72 -0
  32. package/src/constants/categories/profanity.ts +139 -0
  33. package/src/constants/categories/slur.ts +102 -0
  34. package/src/constants/regions/general.ts +111 -2
  35. package/src/constants/regions/jawa.ts +257 -3
  36. package/src/constants/wordList.ts +15 -8
  37. package/src/core/analyzer.ts +28 -13
  38. package/src/core/filter.ts +178 -37
  39. package/src/core/matcher.ts +146 -69
  40. package/src/index.ts +21 -2
  41. package/src/types/index.ts +4 -2
  42. package/src/utils/ahoCorasick.ts +179 -0
  43. package/src/utils/regexUtils.ts +0 -1
  44. package/src/utils/similarityUtils.ts +239 -7
  45. package/tsconfig.json +115 -115
  46. package/.github/workflows/ci.yml +0 -0
  47. package/dist/constants/categories/index.d.ts +0 -9
  48. package/dist/constants/regions/index.d.ts +0 -8
  49. package/src/constants/categories/index.ts +0 -31
  50. package/src/constants/regions/index.ts +0 -62
@@ -14,7 +14,7 @@ export const jawa: ProfanityWord[] = [
14
14
  word: "jancok",
15
15
  category: "sexual",
16
16
  region: "jawa",
17
- severity: 0.8,
17
+ severity: 0.9,
18
18
  aliases: ["jancuk", "jncok", "jancuk", "jncuk", "dancok", "dancuk"],
19
19
  description: "Kata umpatan kasar dalam Bahasa Jawa",
20
20
  context: "Umpatan kasar yang umum digunakan di Jawa Timur",
@@ -52,7 +52,7 @@ export const jawa: ProfanityWord[] = [
52
52
  word: "mbokne ancok",
53
53
  category: "insult",
54
54
  region: "jawa",
55
- severity: 0.8,
55
+ severity: 0.9,
56
56
  aliases: ["mbokne", "mbokneancok"],
57
57
  description: "Umpatan yang menyinggung ibu seseorang",
58
58
  context: "Umpatan kasar yang menyinggung orangtua orang lain",
@@ -61,7 +61,7 @@ export const jawa: ProfanityWord[] = [
61
61
  word: "pekok",
62
62
  category: "insult",
63
63
  region: "jawa",
64
- severity: 0.6,
64
+ severity: 0.7,
65
65
  aliases: ["pekak", "pekilk"],
66
66
  description: "Kata hinaan yang menunjukkan kebodohan",
67
67
  context: "Hinaan untuk menyebut orang yang dianggap sangat bodoh",
@@ -95,6 +95,260 @@ export const jawa: ProfanityWord[] = [
95
95
  context:
96
96
  "Umpatan ringan untuk menanggapi sesuatu yang dianggap tidak benar",
97
97
  },
98
+ {
99
+ word: "kontol",
100
+ category: "sexual",
101
+ region: "jawa",
102
+ severity: 0.8,
103
+ aliases: ["kntl", "kontl"],
104
+ description: "Mengacu ke alat kelamin laki-laki",
105
+ context: "Kata vulgar yang sering digunakan sebagai umpatan kasar",
106
+ },
107
+ {
108
+ word: "tempek",
109
+ category: "sexual",
110
+ region: "jawa",
111
+ severity: 0.8,
112
+ aliases: ["mpek", "torok", "tempk"],
113
+ description: "Mengacu pada alat kelamin perempuan",
114
+ context: "Kata vulgar yang digunakan sebagai umpatan atau hinaan",
115
+ },
116
+ {
117
+ word: "silit",
118
+ category: "insult",
119
+ region: "jawa",
120
+ severity: 0.6,
121
+ aliases: ["selet", "tilis"],
122
+ description: "Mengacu pada bagian dubur atau anus",
123
+ context: "Kata kasar yang digunakan sebagai hinaan",
124
+ },
125
+ {
126
+ word: "mbahmu",
127
+ category: "insult",
128
+ region: "jawa",
129
+ severity: 0.6,
130
+ aliases: ["mbahmu kiper", "mbah mu"],
131
+ description: "Hinaan yang menyinggung nenek/kakek seseorang",
132
+ context: "Umpatan ringan untuk menanggapi sesuatu yang tidak masuk akal",
133
+ },
134
+ {
135
+ word: "makmu",
136
+ category: "insult",
137
+ region: "jawa",
138
+ severity: 0.7,
139
+ aliases: ["mak mu", "mamamu"],
140
+ description: "Hinaan yang menyinggung ibu seseorang",
141
+ context: "Umpatan yang dianggap kasar karena menyinggung orang tua",
142
+ },
143
+ {
144
+ word: "bajingan",
145
+ category: "insult",
146
+ region: "jawa",
147
+ severity: 0.7,
148
+ aliases: [
149
+ "bajilak",
150
+ "bajhingan",
151
+ "bajingak",
152
+ "bajingseng",
153
+ "bajindul",
154
+ "bajigur",
155
+ "jingan",
156
+ ],
157
+ description: "Sebutan untuk orang yang dianggap jahat atau tidak bermoral",
158
+ context: "Umpatan untuk mengekspresikan kemarahan atau kekesalan",
159
+ },
160
+ {
161
+ word: "cocote",
162
+ category: "insult",
163
+ region: "jawa",
164
+ severity: 0.6,
165
+ aliases: ["cocot", "bacot", "nyocot"],
166
+ description: "Mengacu pada mulut dengan konotasi negatif",
167
+ context: "Umpatan untuk menyuruh seseorang berhenti berbicara",
168
+ },
169
+ {
170
+ word: "ngentot",
171
+ category: "sexual",
172
+ region: "jawa",
173
+ severity: 0.9,
174
+ aliases: ["kentu", "kentot", "iclik", "ngtt", "iclk"],
175
+ description: "Mengacu pada aktivitas seksual",
176
+ context: "Kata vulgar yang digunakan sebagai umpatan kasar",
177
+ },
178
+ {
179
+ word: "edan",
180
+ category: "insult",
181
+ region: "jawa",
182
+ severity: 0.5,
183
+ aliases: ["gendeng", "gila", "gendheng", "sarap"],
184
+ description: "Secara harfiah berarti gila atau tidak waras",
185
+ context: "Umpatan untuk menyebut seseorang yang dianggap tidak masuk akal",
186
+ },
187
+ {
188
+ word: "dapuranmu",
189
+ category: "insult",
190
+ region: "jawa",
191
+ severity: 0.6,
192
+ aliases: ["raimu", "rai mu"],
193
+ description: "Secara harfiah mengacu pada wajah atau rupa seseorang",
194
+ context: "Umpatan untuk menghina penampilan atau wajah seseorang",
195
+ },
196
+ {
197
+ word: "damput",
198
+ category: "insult",
199
+ region: "jawa",
200
+ severity: 0.7,
201
+ aliases: ["diamput"],
202
+ description: "Variasi bentuk umpatan dengan makna serupa dengan jancok",
203
+ context: "Umpatan kasar untuk mengekspresikan kemarahan",
204
+ },
205
+ {
206
+ word: "mbathang",
207
+ category: "insult",
208
+ region: "jawa",
209
+ severity: 0.7,
210
+ aliases: ["mbatang"],
211
+ description: "Secara harfiah berarti bangkai",
212
+ context: "Umpatan kasar untuk menghina seseorang",
213
+ },
214
+ {
215
+ word: "ndlogok",
216
+ category: "insult",
217
+ region: "jawa",
218
+ severity: 0.6,
219
+ aliases: ["ndelodok", "ndlodok"],
220
+ description:
221
+ "Mengacu pada tindakan yang dianggap bodoh atau tidak masuk akal",
222
+ context: "Hinaan untuk mengkritik tindakan seseorang",
223
+ },
224
+ {
225
+ word: "nggateli",
226
+ category: "insult",
227
+ region: "jawa",
228
+ severity: 0.5,
229
+ aliases: ["gateli", "gathel"],
230
+ description: "Secara harfiah berarti gatal atau menyebalkan",
231
+ context: "Ungkapan untuk menunjukkan kekesalan terhadap perilaku seseorang",
232
+ },
233
+ {
234
+ word: "perek",
235
+ category: "sexual",
236
+ region: "jawa",
237
+ severity: 0.8,
238
+ aliases: ["lonthe", "pelacur"],
239
+ description: "Istilah merendahkan untuk pekerja seks komersial",
240
+ context: "Kata kasar untuk menghina wanita",
241
+ },
242
+ {
243
+ word: "picek",
244
+ category: "insult",
245
+ region: "jawa",
246
+ severity: 0.6,
247
+ aliases: ["pcek", "buta"],
248
+ description: "Secara harfiah berarti buta atau tidak bisa melihat",
249
+ context: "Hinaan untuk orang yang dianggap tidak bisa melihat kenyataan",
250
+ },
251
+ {
252
+ word: "untumu",
253
+ category: "insult",
254
+ region: "jawa",
255
+ severity: 0.5,
256
+ aliases: ["gigimu", "untu mu"],
257
+ description: "Secara harfiah berarti gigimu",
258
+ context: "Umpatan ringan untuk menanggapi sesuatu yang tidak disetujui",
259
+ },
260
+ {
261
+ word: "goblog",
262
+ category: "insult",
263
+ region: "jawa",
264
+ severity: 0.7,
265
+ aliases: ["ghoblog", "goblok", "gobhlok", "pekok"],
266
+ description: "Kata hinaan yang menunjukkan kebodohan ekstrem",
267
+ context: "Hinaan untuk menyebut seseorang yang dianggap sangat bodoh",
268
+ },
269
+ {
270
+ word: "tolol",
271
+ category: "insult",
272
+ region: "jawa",
273
+ severity: 0.7,
274
+ aliases: ["tholol", "tlol"],
275
+ description: "Kata hinaan yang menunjukkan kebodohan",
276
+ context: "Hinaan untuk menyebut seseorang yang dianggap bodoh",
277
+ },
278
+ {
279
+ word: "budheg",
280
+ category: "insult",
281
+ region: "jawa",
282
+ severity: 0.6,
283
+ aliases: ["budeg", "bdeg"],
284
+ description: "Secara harfiah berarti tuli atau tidak bisa mendengar",
285
+ context: "Hinaan untuk orang yang dianggap tidak mau mendengarkan",
286
+ },
287
+ {
288
+ word: "jiangkrik",
289
+ category: "insult",
290
+ region: "jawa",
291
+ severity: 0.4,
292
+ aliases: ["jiangkrek", "jangkrik"],
293
+ description: "Secara harfiah berarti jangkrik, digunakan sebagai eufemisme",
294
+ context: "Umpatan ringan sebagai pengganti kata kasar yang lebih vulgar",
295
+ },
296
+ {
297
+ word: "diamput",
298
+ category: "insult",
299
+ region: "jawa",
300
+ severity: 0.8,
301
+ aliases: ["damput", "djamput"],
302
+ description: "Bentuk umpatan kasar dengan makna serupa jancok",
303
+ context: "Kata kasar untuk mengekspresikan kemarahan",
304
+ },
305
+ {
306
+ word: "celeng",
307
+ category: "insult",
308
+ region: "jawa",
309
+ severity: 0.6,
310
+ aliases: ["cleng", "babi hutan"],
311
+ description: "Secara harfiah berarti babi hutan",
312
+ context: "Hinaan untuk orang yang dianggap jorok atau rakus",
313
+ },
314
+ {
315
+ word: "kampret",
316
+ category: "insult",
317
+ region: "jawa",
318
+ severity: 0.5,
319
+ aliases: ["kmpret", "kmprt"],
320
+ description: "Secara harfiah berarti kelelawar kecil",
321
+ context: "Umpatan ringan untuk mengekspresikan kekesalan",
322
+ },
323
+ {
324
+ word: "ndeso",
325
+ category: "insult",
326
+ region: "jawa",
327
+ severity: 0.4,
328
+ aliases: ["ndesa", "deso"],
329
+ description: "Secara harfiah berarti dari desa atau kampungan",
330
+ context:
331
+ "Hinaan untuk orang yang dianggap kurang modern atau berpendidikan",
332
+ },
333
+ {
334
+ word: "kere",
335
+ category: "insult",
336
+ region: "jawa",
337
+ severity: 0.5,
338
+ aliases: ["miskin", "mlarat"],
339
+ description: "Secara harfiah berarti miskin atau tidak punya uang",
340
+ context: "Hinaan untuk status ekonomi seseorang yang dianggap rendah",
341
+ },
342
+ {
343
+ word: "itil",
344
+ category: "sexual",
345
+ region: "jawa",
346
+ severity: 0.9,
347
+ aliases: ["itl", "itul"],
348
+ description:
349
+ "Kata vulgar yang mengacu pada bagian dari alat kelamin perempuan",
350
+ context: "Kata vulgar yang merujuk pada anatomi seksual",
351
+ },
98
352
  ];
99
353
 
100
354
  export const jawaWords = jawa.map((item) => item.word);
@@ -1,12 +1,11 @@
1
1
  import { ProfanityWord } from "../types";
2
-
3
- import { sexual, sexualWords } from "./categories/sexual";
4
- import { insult, insultWords } from "./categories/insult";
5
- // import { profanity, profanityWords } from './categories/profanity';
6
- // import { slur, slurWords } from './categories/slur';
7
- // import { drugs, drugsWords } from './categories/drugs';
8
- // import { disgusting, disgustingWords } from './categories/disgusting';
9
- // import { blasphemy, blasphemyWords } from './categories/blasphemy';
2
+ import { sexualWords } from "./categories/sexual";
3
+ import { insultWords } from "./categories/insult";
4
+ // import { profanityWords } from './categories/profanity';
5
+ // import { slurWords } from './categories/slur';
6
+ // import { drugsWords } from './categories/drugs';
7
+ // import { disgustingWords } from './categories/disgusting';
8
+ // import { blasphemyWords } from './categories/blasphemy';
10
9
 
11
10
  import { general, generalWords } from "./regions/general";
12
11
  import { jawa, jawaWords } from "./regions/jawa";
@@ -96,6 +95,14 @@ export const severeWords: string[] = wordObjects
96
95
  * @returns Array dari kata kotor
97
96
  */
98
97
  export function getWordsByFilter(category?: string, region?: string): string[] {
98
+ if (category && !region && category in wordCategories) {
99
+ return wordCategories[category as keyof typeof wordCategories];
100
+ }
101
+
102
+ if (region && !category && region in wordRegions) {
103
+ return wordRegions[region as keyof typeof wordRegions];
104
+ }
105
+
99
106
  return wordObjects
100
107
  .filter((word) => {
101
108
  const matchCategory = category ? word.category === category : true;
@@ -1,7 +1,6 @@
1
1
  import {
2
2
  FilterOptions,
3
3
  AnalysisResult,
4
- ProfanityWord,
5
4
  ProfanityCategory,
6
5
  Region,
7
6
  } from "../types";
@@ -12,13 +11,11 @@ import {
12
11
  findRegions,
13
12
  calculateSeverity,
14
13
  } from "./matcher";
14
+ import { splitIntoSentences } from "../utils/stringUtils";
15
15
  import {
16
- normalizeText,
17
- escapeRegExp,
18
- splitIntoSentences,
19
- getContextAroundIndex,
20
- } from "../utils/stringUtils";
21
- import { findPossibleProfanityBySimiliarity } from "../utils/similarityUtils";
16
+ findPossibleProfanityBySimiliarity,
17
+ findProfanityByLevenshteinDistance,
18
+ } from "../utils/similarityUtils";
22
19
  import { createContextRegex } from "../utils/regexUtils";
23
20
  import { DEFAULT_OPTIONS } from "../config/options";
24
21
 
@@ -60,12 +57,30 @@ export function analyze(
60
57
  }> = [];
61
58
 
62
59
  if (mergedOptions.detectSimilarity) {
63
- const wordList = matchDetails.map((word) => word.word);
64
- similarWords = findPossibleProfanityBySimiliarity(
65
- text,
66
- wordList,
67
- mergedOptions.similarityThreshold || 0.8,
68
- );
60
+ if (matchDetails.length > 0) {
61
+ const wordList = matchDetails.map((word) => word.word);
62
+
63
+ if (mergedOptions.useLevenshtein) {
64
+ const levenshteinResults = findProfanityByLevenshteinDistance(
65
+ text,
66
+ wordList,
67
+ mergedOptions.similarityThreshold || 0.8,
68
+ mergedOptions.maxLevenshteinDistance || 2,
69
+ );
70
+
71
+ similarWords = levenshteinResults.map((item) => ({
72
+ word: item.word,
73
+ original: item.original,
74
+ similarity: item.similarity,
75
+ }));
76
+ } else {
77
+ similarWords = findPossibleProfanityBySimiliarity(
78
+ text,
79
+ wordList,
80
+ mergedOptions.similarityThreshold || 0.8,
81
+ );
82
+ }
83
+ }
69
84
  }
70
85
 
71
86
  return {
@@ -1,12 +1,13 @@
1
1
  import { FilterOptions, FilterResult, ProfanityWord } from "../types";
2
2
  import { findProfanity, findProfanityWithMetadata } from "./matcher";
3
- import { censorWord, escapeRegExp, normalizeText } from "../utils/stringUtils";
3
+ import { censorWord, escapeRegExp } from "../utils/stringUtils";
4
4
  import { createWordRegex } from "../utils/regexUtils";
5
- import {
6
- DEFAULT_OPTIONS,
7
- makeRandomGrawlixString,
8
- getRandomGrawlix,
9
- } from "../config/options";
5
+ import { DEFAULT_OPTIONS, makeRandomGrawlixString } from "../config/options";
6
+
7
+ interface FindProfanityFunction {
8
+ (text: string, options?: FilterOptions): string[];
9
+ lastActualMatches?: Map<string, string[]>;
10
+ }
10
11
 
11
12
  /**
12
13
  * Menyensor kata kotor dalam teks
@@ -28,6 +29,11 @@ export function filter(
28
29
  useRandomGrawlix = false,
29
30
  keepFirstAndLast = false,
30
31
  indonesianVariation = false,
32
+ detectSplit = false,
33
+ detectSimilarity = false,
34
+ useLevenshtein = false,
35
+ maxLevenshteinDistance = 2,
36
+ similarityThreshold = 0.8,
31
37
  } = { ...DEFAULT_OPTIONS, ...options };
32
38
 
33
39
  const matches = findProfanity(text, {
@@ -36,8 +42,16 @@ export function filter(
36
42
  whitelist,
37
43
  checkSubstring,
38
44
  indonesianVariation,
45
+ detectSplit,
46
+ detectSimilarity,
47
+ useLevenshtein,
48
+ maxLevenshteinDistance,
49
+ similarityThreshold,
39
50
  });
40
51
 
52
+ const actualMatches: Map<string, string[]> =
53
+ (findProfanity as FindProfanityFunction).lastActualMatches || new Map();
54
+
41
55
  const matchDetails = findProfanityWithMetadata(text, options);
42
56
 
43
57
  if (matches.length === 0) {
@@ -66,49 +80,176 @@ export function filter(
66
80
  )),
67
81
  );
68
82
 
69
- const regex = createWordRegex(word, {
70
- wholeWord: true,
71
- caseSensitive: false,
72
- leetSpeak: false,
73
- detectSplit: false,
74
- indonesianVariation: false,
75
- });
83
+ const variants = actualMatches.get(word.toLowerCase()) || [];
84
+ variants.push(word);
85
+
86
+ const uniqueVariants = [...new Set(variants)];
76
87
 
77
- let match;
78
- const textToSearch = filteredText;
88
+ uniqueVariants.forEach((variant) => {
89
+ const regex = new RegExp(`\\b${escapeRegExp(variant)}\\b`, "gi");
79
90
 
80
- regex.lastIndex = 0;
91
+ let match;
92
+ while ((match = regex.exec(filteredText)) !== null) {
93
+ const originalWord = match[0];
81
94
 
82
- while ((match = regex.exec(textToSearch)) !== null) {
83
- const originalWord = match[0];
95
+ if (whitelist.includes(originalWord.toLowerCase())) continue;
84
96
 
85
- if (whitelist.includes(originalWord.toLowerCase())) continue;
97
+ let censoredWord;
98
+ if (useRandomGrawlix) {
99
+ censoredWord = makeRandomGrawlixString(originalWord.length);
100
+ } else {
101
+ censoredWord = censorWord(
102
+ originalWord,
103
+ replaceWith,
104
+ !fullWordCensor && keepFirstAndLast,
105
+ );
106
+ }
86
107
 
87
- let censoredWord;
88
- if (useRandomGrawlix) {
89
- censoredWord = makeRandomGrawlixString(originalWord.length);
90
- } else {
91
- censoredWord = censorWord(
92
- originalWord,
93
- replaceWith,
94
- !fullWordCensor && keepFirstAndLast,
108
+ replacements.push({
109
+ original: originalWord,
110
+ censored: censoredWord,
111
+ metadata,
112
+ });
113
+
114
+ filteredText = filteredText.replace(
115
+ new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"),
116
+ censoredWord,
95
117
  );
96
118
  }
119
+ });
97
120
 
98
- replacements.push({
99
- original: originalWord,
100
- censored: censoredWord,
101
- metadata,
102
- });
121
+ if (detectSplit || detectLeetSpeak) {
122
+ if (detectLeetSpeak) {
123
+ const leetRegex = createWordRegex(word, {
124
+ wholeWord: true,
125
+ caseSensitive: false,
126
+ leetSpeak: true,
127
+ detectSplit: false,
128
+ indonesianVariation: false,
129
+ });
103
130
 
104
- const replaceRegex = new RegExp(
105
- `\\b${escapeRegExp(originalWord)}\\b`,
106
- "g",
107
- );
108
- filteredText = filteredText.replace(replaceRegex, censoredWord);
131
+ let match;
132
+ while ((match = leetRegex.exec(filteredText)) !== null) {
133
+ const originalWord = match[0];
134
+
135
+ if (whitelist.includes(originalWord.toLowerCase())) continue;
136
+
137
+ let censoredWord;
138
+ if (useRandomGrawlix) {
139
+ censoredWord = makeRandomGrawlixString(originalWord.length);
140
+ } else {
141
+ censoredWord = censorWord(
142
+ originalWord,
143
+ replaceWith,
144
+ !fullWordCensor && keepFirstAndLast,
145
+ );
146
+ }
147
+
148
+ replacements.push({
149
+ original: originalWord,
150
+ censored: censoredWord,
151
+ metadata,
152
+ });
153
+
154
+ filteredText = filteredText.replace(
155
+ new RegExp(escapeRegExp(originalWord), "g"),
156
+ censoredWord,
157
+ );
158
+ }
159
+ }
160
+
161
+ if (detectSplit) {
162
+ const splitRegex = createWordRegex(word, {
163
+ wholeWord: false,
164
+ caseSensitive: false,
165
+ leetSpeak: false,
166
+ detectSplit: true,
167
+ indonesianVariation: false,
168
+ });
169
+
170
+ let match;
171
+ while ((match = splitRegex.exec(filteredText)) !== null) {
172
+ const originalWord = match[0];
173
+
174
+ if (whitelist.includes(originalWord.toLowerCase())) continue;
175
+
176
+ let censoredWord;
177
+ if (useRandomGrawlix) {
178
+ censoredWord = makeRandomGrawlixString(originalWord.length);
179
+ } else {
180
+ censoredWord = censorWord(
181
+ originalWord,
182
+ replaceWith,
183
+ !fullWordCensor && keepFirstAndLast,
184
+ );
185
+ }
186
+
187
+ replacements.push({
188
+ original: originalWord,
189
+ censored: censoredWord,
190
+ metadata,
191
+ });
192
+
193
+ filteredText = filteredText.replace(
194
+ new RegExp(escapeRegExp(originalWord), "g"),
195
+ censoredWord,
196
+ );
197
+ }
198
+ }
109
199
  }
110
200
  });
111
201
 
202
+ if (detectSimilarity && useLevenshtein) {
203
+ matches.forEach((word) => {
204
+ const metadata = matchDetails.find(
205
+ (m) =>
206
+ m.word.toLowerCase() === word.toLowerCase() ||
207
+ (m.aliases &&
208
+ m.aliases.some(
209
+ (alias) => alias.toLowerCase() === word.toLowerCase(),
210
+ )),
211
+ );
212
+
213
+ const variants = actualMatches.get(word.toLowerCase()) || [];
214
+
215
+ variants.forEach((variant) => {
216
+ const exactVariantRegex = new RegExp(
217
+ `\\b${escapeRegExp(variant)}\\b`,
218
+ "gi",
219
+ );
220
+
221
+ let match;
222
+ while ((match = exactVariantRegex.exec(filteredText)) !== null) {
223
+ const originalWord = match[0];
224
+
225
+ if (whitelist.includes(originalWord.toLowerCase())) continue;
226
+
227
+ let censoredWord;
228
+ if (useRandomGrawlix) {
229
+ censoredWord = makeRandomGrawlixString(originalWord.length);
230
+ } else {
231
+ censoredWord = censorWord(
232
+ originalWord,
233
+ replaceWith,
234
+ !fullWordCensor && keepFirstAndLast,
235
+ );
236
+ }
237
+
238
+ replacements.push({
239
+ original: originalWord,
240
+ censored: censoredWord,
241
+ metadata,
242
+ });
243
+
244
+ filteredText = filteredText.replace(
245
+ new RegExp(`\\b${escapeRegExp(originalWord)}\\b`, "g"),
246
+ censoredWord,
247
+ );
248
+ }
249
+ });
250
+ });
251
+ }
252
+
112
253
  return {
113
254
  filtered: filteredText,
114
255
  censored: replacements.length,