@privacyscrubber/mcp-server 1.7.6 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.well-known/mcp/server-card.json +78 -0
- package/LICENSE +21 -0
- package/README.md +5 -5
- package/index.js +168 -9
- package/package.json +1 -1
- package/ps-license-manager.js +10 -16
- package/ps-pii-engine.cjs +472 -116
- package/ps-pii-engine.js +472 -116
- package/scrubber-core.cjs +438 -150
package/ps-pii-engine.cjs
CHANGED
|
@@ -19,7 +19,9 @@ let DEVOPS_SECRETS = [
|
|
|
19
19
|
{ name: 'JSON Web Token (JWT)', type: 'SECRET', regex: /\beyJ[a-zA-Z0-9_-]+\.[a-zA-Z0-9_-]+\.[a-zA-Z0-9_-]+\b/g },
|
|
20
20
|
{ name: 'API Token/Key (GitHub/Slack/NPM)', type: 'SECRET', regex: /\b(?:ghp|gho|ghu|ghs|ghr|glpat|npm|xox[baprs])[-_][A-Za-z0-9_]{10,}\b/g },
|
|
21
21
|
{ name: 'Stripe API Key', type: 'SECRET', regex: /\b(?:[rs]k)_(?:test|live)_[a-zA-Z0-9]{24,}\b/g },
|
|
22
|
-
{ name: '
|
|
22
|
+
{ name: 'OpenAI Project API Key', type: 'SECRET', regex: /\b(?:sk|pk)-(?:proj-)?[a-zA-Z0-9_-]{16,}\b/gi },
|
|
23
|
+
{ name: 'Database Connection URI', type: 'SECRET', regex: /\b(?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp|mssql):\/\/[^\s"']+/gi },
|
|
24
|
+
{ name: 'Generic Secret/Key', type: 'SECRET', regex: /\b(sk|pk|secret|key|token|auth)(?:[-_][a-zA-Z0-9_-]{3,}|(?=[a-zA-Z0-9_-]{5,}\b)(?=[a-zA-Z_-]*[0-9])[a-zA-Z0-9_-]{5,})\b/gi },
|
|
23
25
|
{ name: 'Hash / Hex Key (32-64 chars)', type: 'SECRET', regex: /\b[a-fA-F0-9]{32,64}\b/g },
|
|
24
26
|
{ name: 'CVE Identifier', type: 'SECRET', regex: /\bCVE-\d{4}-\d{4,}\b/gi },
|
|
25
27
|
{ name: 'Cryptographic Hash', type: 'SECRET', regex: /\b(MD5|SHA1|SHA256)[:\s][a-f0-9]{32,64}\b/gi },
|
|
@@ -31,25 +33,29 @@ let DEVOPS_SECRETS = [
|
|
|
31
33
|
let REGEX_RULES = [
|
|
32
34
|
...DEVOPS_SECRETS,
|
|
33
35
|
// Emails
|
|
34
|
-
{ type: 'EMAIL', regex:
|
|
36
|
+
{ type: 'EMAIL', regex: /\b[a-zA-Z0-9._%+-]{1,64}@[a-zA-Z0-9.-]{1,255}\.[a-zA-Z]{2,}\b/g },
|
|
35
37
|
|
|
36
|
-
// Financial Data
|
|
37
|
-
{ type: 'FINANCIAL', regex:
|
|
38
|
-
{ type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9]([A-Z0-9]{3})?\b/g },
|
|
39
|
-
{ type: 'FINANCIAL', regex: /\b\d{9}\b/g },
|
|
38
|
+
// Financial Data (PCI-DSS, Bank Accounts, Direct Deposits, Cards, IBAN, SWIFT, Routing Numbers)
|
|
39
|
+
{ type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9](?:[A-Z0-9]{3})?\b/g },
|
|
40
40
|
{ type: 'FINANCIAL', regex: /\b[A-Z]{2}[0-9]{2}[a-zA-Z0-9]{4}[0-9]{7}[a-zA-Z0-9]{0,16}\b/g },
|
|
41
41
|
{ type: 'FINANCIAL', regex: /\bPORTFOLIO[-_][A-Z0-9]{5,}\b/gi },
|
|
42
42
|
{ type: 'FINANCIAL', regex: /\b(?:\d[ -]?){13,19}\b/g },
|
|
43
43
|
{ type: 'FINANCIAL', regex: /\b(?:1|3|bc1)[a-zA-HJ-NP-Z0-9]{25,39}\b/g },
|
|
44
44
|
{ type: 'FINANCIAL', regex: /\b0x[a-fA-F0-9]{40}\b/g },
|
|
45
|
+
// Masked / Direct Deposit Bank Account Numbers & Routing Numbers
|
|
46
|
+
{ type: 'FINANCIAL', regex: /\b(?:Account|Acct|Checking|Savings|Direct\s+Deposit)\s*(?:#|ID|No\.?|Number)?[:\s#]*(?:[\*xX•.-]{3,}\d{2,6}|\d{4}[-\s]?\d{4}[-\s]?\d{2,6})\b/gi },
|
|
47
|
+
{ type: 'FINANCIAL', regex: /\b(?:ABA|Routing|RTN)\s*(?:#|ID|No\.?|Number)?[:\s#]*\d{9}\b/gi },
|
|
45
48
|
|
|
46
49
|
// Legal & Court
|
|
47
|
-
{ type: 'LEGAL', regex: /\bCASE[-_][A-Z0-
|
|
48
|
-
{ type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-
|
|
49
|
-
{ type: 'LEGAL', regex: /\b[A-Z]{2,4}[- ]?\d{2}[- ]?\d{4,}\b/g },
|
|
50
|
+
{ type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9_-]{4,}\b/gi },
|
|
51
|
+
{ type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9_-]{4,}\b/gi },
|
|
50
52
|
{ type: 'PRIVILEGE', regex: /ATTORNEY[- ]CLIENT[- ]PRIVILEGE/gi },
|
|
51
53
|
|
|
52
|
-
// Professional IDs
|
|
54
|
+
// Professional IDs & Organizations
|
|
55
|
+
{ type: 'NAME', isContextName: true, regex: /\b(?:[A-Z][A-Za-z0-9&.,'-]*[ \t\xA0]+){1,5}(?:Inc\.?|LLC|Corp\.?|Corporation|Ltd\.?|Limited|Co\.?|Company|Group|Holdings|Solutions|Services|Technologies|Logistics|Industries|Capital|Bank|Partners|LLP|PLLC)(?:\s+(?:LLC|Inc\.?|Corp\.?|Ltd\.?|USA|Group))?\b/g },
|
|
56
|
+
{ type: 'ID', regex: /\b(?:Employee|Emp|EE|Worker|Staff|File|Badge|Member|Advisor|Producer|Agent|Borrower)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Z0-9-]{3,15}\b/gi },
|
|
57
|
+
{ type: 'ID', regex: /\b(?:Pay\s+Group|Cost\s+Center|Dept|Department)[:\s#]*[A-Za-z0-9_-]{2,30}\b/gi },
|
|
58
|
+
{ type: 'ID', regex: /(?:\b(?:Box\s+d\b|d\.\s*(?:Control|#)?|d\s+Control)\s*(?:number|no\.?|#|num)?[:\s#]*|\bControl\s*(?:number|no\.?|#|num)[:\s#]*|\bControl[:#]\s*)([A-Za-z0-9-]{3,30})/gi },
|
|
53
59
|
{ type: 'ID', regex: /\bEEID[ -]?\d{4,}\b/gi },
|
|
54
60
|
{ type: 'ID', regex: /\bRESUME[-_]?[A-Z0-9]{4,}\b/gi },
|
|
55
61
|
{ type: 'ID', regex: /\bLEAD[-_][A-Z0-9]{5,}\b/gi },
|
|
@@ -69,34 +75,45 @@ let REGEX_RULES = [
|
|
|
69
75
|
{ type: 'ID', regex: /\bCOURSE[-_][A-Z0-9]{4,}\b/gi },
|
|
70
76
|
{ type: 'ID', regex: /\bINSTANCE[-_]ID[-_][a-z0-9-]{10,}\b/gi },
|
|
71
77
|
{ type: 'ID', regex: /\bENV[-_][A-Z0-9]{3,}\b/gi },
|
|
72
|
-
{ type: '
|
|
78
|
+
{ type: 'ID', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
|
|
73
79
|
|
|
74
|
-
//
|
|
75
|
-
{ type: '
|
|
76
|
-
{ type: '
|
|
80
|
+
// Insurance & Health Plan IDs
|
|
81
|
+
{ type: 'ID', regex: /\b(?:BCBS|AETNA|CIGNA|UHC|HUMANA|MEDICARE|MEDICAID)[-_A-Za-z0-9]+\b/gi },
|
|
82
|
+
{ type: 'ID', regex: /\b(?:Insurance|Policy|Member|Subscriber|Group|Plan|Health|Rx)[-_: ]*ID[:\s#]*([A-Za-z0-9-]+)/gi },
|
|
77
83
|
|
|
84
|
+
// Addresses & Locations
|
|
85
|
+
{ type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:\bBox\s+f\b|\bf\.\s*|\bf\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?(?:address[,\s]+and\s+ZIP\s+code|address)?|(?:Employee(?:'s)?\s+address[,\s]+and\s+ZIP\s+code))[\s:#]*([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|Box|\d+\b|1\b|2\b|Wages|Federal|Social|Medicare)/gi },
|
|
86
|
+
{ type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:Borrower(?:'s)?|Co-Borrower(?:'s)?|Employee(?:'s)?|Employer(?:'s)?|Home|Mailing|Property|Physical)\s+address)[\s:#]+([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|City|State|ZIP|SSN|EIN|Phone|Box|\d+\b)/gi },
|
|
87
|
+
{ type: 'ADDRESS', regex: /\b\d{1,6}[ \t\xA0]+(?:[A-Za-z0-9.-]+[ \t\xA0]+){1,4}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Hwy|Highway|Cir|Circle|Trl|Trail|Loop|Row|Pike|Box|PO Box|P\.O\.[ \t\xA0]*Box)\b(?:[ \t\xA0]*,?[ \t\xA0]*(?:Apt|Apartment|Suite|Ste|Unit|#|Fl|Floor|Bldg|Building)\.?[ \t\xA0]*[A-Za-z0-9-]+)?/gi },
|
|
88
|
+
{ type: 'ADDRESS', regex: /\b(?:P\.?O\.?[ \t\xA0]*Box|PO[ \t\xA0]*Box)[ \t\xA0]+\d{1,6}\b/gi },
|
|
89
|
+
{ type: 'ADDRESS', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
|
|
90
|
+
{ type: 'ADDRESS', regex: /\b(?:ZIP|Postal|Code)?\s*(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
|
|
91
|
+
{ type: 'ADDRESS', regex: /\b\d{5}-\d{4}\b/g },
|
|
92
|
+
{ type: 'LOCATION', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\b/g },
|
|
78
93
|
|
|
79
94
|
// PHI & Medical
|
|
95
|
+
{ type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record No|Patient No)[\s:#]+([A-Za-z0-9-]+)/gi },
|
|
80
96
|
{ type: 'PHI', regex: /\bMRN[ -]?\d{6,}\b/gi },
|
|
81
97
|
{ type: 'PHI', regex: /\b[A-TV-Z]\d{2}[. ]?\d[A-Z0-9]?\b/g },
|
|
82
98
|
{ type: 'PHI', regex: /\b[A-Z]{2,3}\d{6,8}\b/g },
|
|
83
99
|
{ type: 'PHI', regex: /\bNHS[ -]?\d{3}[ -]?\d{3}[ -]?\d{4}\b/gi },
|
|
84
100
|
|
|
85
|
-
|
|
86
101
|
// Copyrights
|
|
87
102
|
{ type: 'COPYRIGHT', regex: /\bPROJECT[-_][A-Z0-9]{5,}\b/gi },
|
|
88
103
|
{ type: 'COPYRIGHT', regex: /\b(DRAFT|ASSET|SCRIPT)[-_][0-9]{4,}\b/gi },
|
|
89
104
|
|
|
90
|
-
// General Privacy &
|
|
91
|
-
{ type: '
|
|
92
|
-
{ type: '
|
|
93
|
-
{ type: '
|
|
94
|
-
{ type: '
|
|
95
|
-
{ type: '
|
|
105
|
+
// General Privacy, Dates & Secrets
|
|
106
|
+
{ type: 'ID', regex: /\b(?:GDPR|HIPAA|CCPA|SOC2)[-_]AUDIT[-_]\d{4}\b/gi },
|
|
107
|
+
{ type: 'ID', regex: /\bPOLICY[-_][A-Z0-9]{5,}\b/gi },
|
|
108
|
+
{ type: 'ID', regex: /\bGRADE[S]?\s*:\s*[A-DF][+-]?\b/gi },
|
|
109
|
+
{ type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
|
|
110
|
+
{ type: 'SECRET', regex: /\b(?:PASSWORD|PWD|SECRET)\s*[:=]\s*["']?[\S]{4,}["']?/gi },
|
|
96
111
|
|
|
97
112
|
// Standard IDs (SSN, EIN, Passport, VAT)
|
|
98
113
|
{ type: 'ID', regex: /\b\d{3}-\d{2}-\d{4}\b/g },
|
|
99
|
-
{ type: 'ID', regex:
|
|
114
|
+
{ type: 'ID', regex: /\b(?:XXX|xxx|\*\*\*)[ -]?(?:XX|xx|\*\*)[ -]?\d{4}\b/g },
|
|
115
|
+
{ type: 'ID', regex: /\b\d{2}-\d{7}\b/g },
|
|
116
|
+
{ type: 'ID', regex: /(?:(?:[A-Za-z]:\\|\/(?:usr|var|etc|home|root|Users|private|tmp|opt|bin|sbin|dev|Applications|Library)\/)[a-zA-Z0-9_.-]+(?:[\/\\][a-zA-Z0-9_.-]+)*|\/(?:[a-zA-Z0-9_.-]+\/)+[a-zA-Z0-9_.-]+\.(?:txt|pdf|docx|xlsx|csv|js|ts|json|env|log|key|pem|crt|conf|yaml|yml|xml|html|sql|py|go|rs|c|cpp|h|sh|bin|zip|tar|gz|png|jpg|jpeg|svg|webp|wasm)\b)/g },
|
|
100
117
|
{ type: 'ID', regex: /\b[A-CEGHJ-PR-TW-Z]{1}[A-CEGHJ-NPR-TW-Z]{1}[0-9]{6}[A-DFM]{1}\b/gi },
|
|
101
118
|
{ type: 'ID', regex: /\b[A-Z]{2}[0-9]{6,12}\b/gi },
|
|
102
119
|
{ type: 'ID', regex: /[A-Z0-9<]{30,44}/g },
|
|
@@ -105,14 +122,13 @@ let REGEX_RULES = [
|
|
|
105
122
|
{ type: 'IP', regex: /\b(?:\d{1,3}\.){3}\d{1,3}\b/g },
|
|
106
123
|
{ type: 'IP', regex: /\b(?:[a-fA-F0-9]{1,4}:){7}[a-fA-F0-9]{1,4}\b/g },
|
|
107
124
|
{ type: 'ID', regex: /\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b/g },
|
|
108
|
-
{ type: 'ID', regex: /(?:\B\/|\b[a-zA-Z]:\\)(?:[\w.-]+[\/\\]
|
|
125
|
+
{ type: 'ID', regex: /(?:\B\/|\b[a-zA-Z]:\\)(?:[\w.-]+[\/\\])*[\w.-]+\b/g },
|
|
109
126
|
{ type: 'ID', regex: /\b\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}(?::\d{2})?)?\b/g },
|
|
110
127
|
{ type: 'ID', regex: /\b\d{4}-\d{2}-\d{2}\b/g },
|
|
111
128
|
{ type: 'ID', regex: /\b\d{2}\/\d{2}\/\d{4}\b/g },
|
|
112
129
|
|
|
113
|
-
// Phone Numbers
|
|
114
|
-
{ type: 'PHONE', regex:
|
|
115
|
-
{ type: 'PHONE', regex: /(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}/g },
|
|
130
|
+
// Phone Numbers
|
|
131
|
+
{ type: 'PHONE', regex: /(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}\b/g },
|
|
116
132
|
{ type: 'PHONE', regex: /\+?[1-9]\d{1,3}[\s.-]\(?\d{1,4}\)?[\s.-]\d{2,4}[\s.-]\d{4}/g },
|
|
117
133
|
{ type: 'PHONE', regex: /(?:\+44\s?7\d{3}|\(?07\d{3}\)?)\s?\d{3}\s?\d{3}\b/g },
|
|
118
134
|
{ type: 'PHONE', regex: /\b(?:\d{3}[-.\s]\d{4}|\(\d{3}\)\s??\d{3}[-.\s]??\d{4}|\d{3}[-.\s]??\d{3}[-.\s]??\d{4})\b/g },
|
|
@@ -126,50 +142,63 @@ let REGEX_RULES = [
|
|
|
126
142
|
{ type: 'ID', regex: /\bDRIVER[S]?\s+LICENSE[ -]?\d{6,15}\b/gi },
|
|
127
143
|
|
|
128
144
|
// Generalized Name Detection (First [Middle] Last) — max 1 middle word to avoid grabbing job titles
|
|
129
|
-
// Middle word must be: a particle (van/de/etc.), a single initial (A.), or a capitalized word of ≥2 lowercase letters
|
|
130
|
-
{ type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}[\p{Ll}'-]*
|
|
131
|
-
//
|
|
132
|
-
{ type: 'NAME', regex:
|
|
133
|
-
//
|
|
134
|
-
|
|
135
|
-
|
|
145
|
+
// Middle word must be: a particle (van/de/etc.), a single initial (A. or A), or a capitalized word of ≥2 lowercase letters
|
|
146
|
+
{ type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*|\p{Lu}\.?|van|von|de|di|da|la|le|del|du|der|van[ \t\xA0]+de|van[ \t\xA0]+der)){0,1}[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
147
|
+
// Payroll Format Names (e.g. "BARKER, KELLY", "BARKER, KELLY M", "DOE, JOHN M.")
|
|
148
|
+
{ type: 'NAME', regex: /\b[A-Z]{2,25},\s+[A-Z]{2,25}(?:\s+[A-Z]\.?|\s+[A-Z]{2,25})*\b/g },
|
|
149
|
+
// All-Caps Names (2–3 words, supporting single-letter middle initial e.g. "KELLY M BARKER", "JOHN M BARKER")
|
|
150
|
+
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+(?:\p{Lu}\.?|[A-Z][a-z]+))?[ \t\xA0]+\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
151
|
+
// ALL-CAPS first + middle initial(s) with optional spaces + ALL-CAPS last name
|
|
152
|
+
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
136
153
|
// ALL-CAPS first name + optional middle initial(s) with optional spaces + Mixed-Case last name
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])(?:Mr|Mrs|Ms|Dr|Prof|Hon|Mr\.|Mrs\.|Ms\.|Dr\.|Prof\.|Hon\.)[ \t\xA0]+\p{Lu}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
154
|
+
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*|[ \t\xA0]+)(?:\p{Lu}\p{Ll}[\p{Ll}'-]*|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
155
|
+
// Names with Honorifics (with or without period, supporting single or multi-word full names) (Unicode-safe)
|
|
156
|
+
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])(?:Mr|Mrs|Ms|Dr|Prof|Hon|Mr\.|Mrs\.|Ms\.|Dr\.|Prof\.|Hon\.)[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*))?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
141
157
|
|
|
142
|
-
//
|
|
143
|
-
{ type: 'NAME', isContextName: true, regex: /(?:
|
|
158
|
+
// W-2 Box c Employer Block (Name, Address, and Zip Code)
|
|
159
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:\bBox\s+c\b|\bc\.\s*|\bc\s+(?=Employer))\s*(?:Employer(?:'s)?\s*)?(?:name[,\s]+address[,\s]+and\s+ZIP\s+code|name)?|(?:Employer(?:'s)?\s+name[,\s]+address[,\s]+and\s+ZIP\s+code))[\s:#]*([A-Za-z0-9&., \t\xA0'-]{2,45}?)(?=\r?\n|$|\s{3,}|\t|EIN|FEIN|Box|\d+\b|Wages|Federal|Social|Medicare)/gi },
|
|
160
|
+
// W-2 Box e Employee Name & Initial
|
|
161
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:\bBox\s+e\b|\be\.\s*|\be\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?(?:first\s+name(?:\s+(?:and|&)\s+initial)?|name)?[:\s#]*|(?:Employee(?:'s)?\s+first\s+name(?:\s+(?:and|&)\s+initial)?))[\s:#]*([A-Za-z0-9.\s'-]{2,35}?)(?=\r?\n|$|\s{3,}|\t|Last|Surname|Suff|Box|\d+\b|1\b)/gi },
|
|
162
|
+
// Contextual First Names (Employee's first name, First name, Given name)
|
|
163
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:Employee(?:'s)?|Borrower(?:'s)?|Co-Borrower(?:'s)?|Applicant(?:'s)?|Candidate(?:'s)?|Worker(?:'s)?|Taxpayer(?:'s)?|Spouse(?:'s)?|Person(?:'s)?)\s+)?(?:First\s+name(?:\s+(?:and|&)\s+initial)?|Given\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9.\s'-]{2,30}?)(?=\r?\n|$|\s{3,}|\t|Last|Surname|Family|Suff|Box|Address|SSN|EIN)/gi },
|
|
164
|
+
// Contextual Last Names (Last name, Surname, Family name)
|
|
165
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:Employee(?:'s)?|Borrower(?:'s)?|Co-Borrower(?:'s)?|Applicant(?:'s)?|Candidate(?:'s)?|Worker(?:'s)?|Taxpayer(?:'s)?|Spouse(?:'s)?|Person(?:'s)?)\s+)?(?:Last\s+name|Surname|Family\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z'-]{2,30})/gi },
|
|
166
|
+
// Contextual General Names (Employee, Borrower, Co-Borrower, Taxpayer, Spouse, Applicant, Candidate, Worker, Employer, Company, Insured, Patient, Client, etc.)
|
|
167
|
+
{ type: 'NAME', isContextName: true, regex: /(?:Employee(?:\s+Name)?|Employer(?:\s+Name)?|Borrower(?:\s+Name)?|Co-Borrower(?:\s+Name)?|Applicant(?:\s+Name)?|Candidate(?:\s+Name)?|Worker(?:\s+Name)?|Taxpayer(?:\s+Name)?|Spouse(?:\s+Name)?|Manager|Supervisor|Reporting To|Insured|Claimant|Patient|Client|Customer|Account Holder|Prepared By|Attention|Attn|Contact(?: Name)?|Child|Parent|Guardian|Relationship|Kin|Tenant|Landlord|Buyer|Seller|Plaintiff|Defendant|Testator)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9&.,\s'-]{2,40}?)(?=\r?\n|$|\s{3,}|\t|Employee|Employer|Address|Phone|SSN|EIN|FEIN|Date|Pay|Rate|Tax|W-2|OMB|Copy|Box|Status)/gi },
|
|
168
|
+
// Box e shorthand
|
|
169
|
+
{ type: 'NAME', isContextName: true, regex: /\b(?:Box\s+e)\s*[:#-]\s*([A-Za-z0-9&.,\s'-]{2,40})/gi }
|
|
144
170
|
];
|
|
145
171
|
|
|
146
172
|
let PROFILE_RULES = {
|
|
147
173
|
general: [],
|
|
148
174
|
legal: [
|
|
149
|
-
{ type: 'LEGAL', regex: /\bCASE[-_][A-Z0-
|
|
150
|
-
{ type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-
|
|
175
|
+
{ type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9_-]{4,}\b/gi },
|
|
176
|
+
{ type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9_-]{4,}\b/gi },
|
|
151
177
|
{ type: 'LEGAL', regex: /\b[A-Z]{2,4}[- ]?\d{2}[- ]?\d{4,}\b/g },
|
|
152
178
|
{ type: 'PRIVILEGE', regex: /ATTORNEY[- ]CLIENT[- ]PRIVILEGE/gi }
|
|
153
179
|
],
|
|
154
180
|
hr: [
|
|
155
|
-
{ type: 'NAME', isContextName: true, regex: /(?:Candidate|Applicant|Employee|Reporting To|Manager)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g },
|
|
156
|
-
|
|
181
|
+
{ type: 'NAME', isContextName: true, regex: /(?:Candidate|Applicant|Employee|Reporting To|Manager|Mentored by|Direct Report)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g },
|
|
157
182
|
{ type: 'ID', regex: /\bEEID[ -]?\d{4,}\b/gi },
|
|
158
183
|
{ type: 'ID', regex: /\bEMP[-_]\d{3,}\b/gi },
|
|
159
184
|
{ type: 'ID', regex: /\bRESUME[-_]?[A-Z0-9]{4,}\b/gi },
|
|
160
|
-
{ type: '
|
|
161
|
-
{ type: '
|
|
185
|
+
{ type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
|
|
186
|
+
{ type: 'DATE', regex: /\b(?:Graduated|Graduation|Class of)[:\s]+(?:(?:Spring|Summer|Fall|Winter)\s+)?(?:(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec|January|February|March|April|June|July|August|September|October|November|December)\s+)?\d{4}\b/gi },
|
|
187
|
+
{ type: 'ID', regex: /(?:https?:\/\/)?(?:www\.)?linkedin\.com\/in\/[A-Za-z0-9_-]+/gi },
|
|
188
|
+
{ type: 'ID', regex: /(?:https?:\/\/)?(?:www\.)?github\.com\/[A-Za-z0-9_-]+/gi },
|
|
189
|
+
{ type: 'ADDRESS', regex: /\b\d{1,6}\s+(?:[A-Z0-9][a-zA-Z0-9-]*\s+){1,3}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Highway|Hwy|Circle|Cir|Trail|Trl|Dr(?!\.?\s+[A-Z][a-z]+))\b/g }
|
|
162
190
|
],
|
|
163
191
|
finance: [
|
|
164
|
-
{ type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9]([A-Z0-9]{3})?\b/g },
|
|
192
|
+
{ type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9](?:[A-Z0-9]{3})?\b/g },
|
|
165
193
|
{ type: 'FINANCIAL', regex: /\b[A-Z]{2}[0-9]{2}[a-zA-Z0-9]{4}[0-9]{7}[a-zA-Z0-9]{0,16}\b/g },
|
|
166
194
|
{ type: 'FINANCIAL', regex: /\bPORTFOLIO[-_][A-Z0-9]{5,}\b/gi }
|
|
167
195
|
],
|
|
168
196
|
medical: [
|
|
169
|
-
{ type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record No|Patient No)[\s:#]
|
|
170
|
-
{ type: '
|
|
171
|
-
|
|
197
|
+
{ type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record No|Patient No)[\s:#]+([A-Za-z0-9-]+)/gi },
|
|
198
|
+
{ type: 'DATE', regex: /\b(?:DOB|Date of Birth|BIRTHDAY)[\s:]+([0-9./-]{6,10})\b/gi },
|
|
172
199
|
{ type: 'PHI', regex: /\bMRN[ -]?\d{6,}\b/gi },
|
|
200
|
+
{ type: 'ID', regex: /\b(?:Insurance|Policy|Member|Subscriber|Group|Plan|Health|Rx)[-_: ]*ID[:\s#]*([A-Za-z0-9-]+)/gi },
|
|
201
|
+
{ type: 'ID', regex: /\b(?:BCBS|AETNA|CIGNA|UHC|HUMANA|MEDICARE|MEDICAID)[-_A-Za-z0-9]+\b/gi },
|
|
173
202
|
{ type: 'PHI', regex: /\b[A-TV-Z]\d{2}[. ]?\d[A-Z0-9]?\b/g },
|
|
174
203
|
{ type: 'PHI', regex: /\b[A-Z]{2,3}\d{6,8}\b/g },
|
|
175
204
|
{ type: 'PHI', regex: /\bNHS[ -]?\d{3}[ -]?\d{3}[ -]?\d{4}\b/gi }
|
|
@@ -186,13 +215,13 @@ let PROFILE_RULES = {
|
|
|
186
215
|
bizops: [
|
|
187
216
|
{ type: 'ID', regex: /\b(?:DEAL|KPI|METRIC)[-_: ]?[A-Z0-9]{4,}\b/gi },
|
|
188
217
|
{ type: 'ID', regex: /\b(?:ENTITY|VENDOR|PARTNER)[-_: ]?[0-9]{4,10}\b/gi },
|
|
189
|
-
{ type: 'FINANCIAL', regex: /\b(?:REVENUE|EBITDA|PROFIT|MARGIN)[-
|
|
218
|
+
{ type: 'FINANCIAL', regex: /\b(?:REVENUE|EBITDA|PROFIT|MARGIN)[\s:_-]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
|
|
190
219
|
{ type: 'SECRET', regex: /\b(?:NDA|M&A|MERGER)[-_: ]?[A-Z0-9]{4,}\b/gi }
|
|
191
220
|
],
|
|
192
221
|
sales: [
|
|
193
222
|
{ type: 'ID', regex: /\bOPPORTUNITY[-_: ]?[A-Z0-9]{5,}\b/gi },
|
|
194
223
|
{ type: 'ID', regex: /\b(?:DOCUSIGN|CONTRACT)[-_: ]?[0-9A-F]{8,32}\b/gi },
|
|
195
|
-
{ type: 'FINANCIAL', regex: /\b(?:ARR|MRR|QUOTA)[\s:]
|
|
224
|
+
{ type: 'FINANCIAL', regex: /\b(?:ARR|MRR|QUOTA)[\s:]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
|
|
196
225
|
{ type: 'ID', regex: /\b(?:SFDC|HUBSPOT)[-_: ]?[0-9A-Z]{15,18}\b/gi }
|
|
197
226
|
],
|
|
198
227
|
support: [
|
|
@@ -205,8 +234,8 @@ let PROFILE_RULES = {
|
|
|
205
234
|
{ type: 'ID', regex: /\b(?:MLS|LIS)[- ]?\d{6,10}\b/gi },
|
|
206
235
|
{ type: 'ID', regex: /\bPARCEL[- ]?\d{5,15}\b/gi },
|
|
207
236
|
{ type: 'ID', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
|
|
208
|
-
{ type: 'FINANCIAL', regex: /\b(?:RENT|LEASE|ESCROW)[\s:]
|
|
209
|
-
{ type: 'SECRET', regex: /\b(?:GATE|DOOR|LOBBY)[-_ ](?:CODE|PIN)[\s:]
|
|
237
|
+
{ type: 'FINANCIAL', regex: /\b(?:RENT|LEASE|ESCROW)[\s:]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)[0-9,]{3,}\b/gi },
|
|
238
|
+
{ type: 'SECRET', regex: /\b(?:GATE|DOOR|LOBBY)[-_ ](?:CODE|PIN)[\s:]*\d{4,6}\b/gi }
|
|
210
239
|
],
|
|
211
240
|
compliance: [
|
|
212
241
|
{ type: 'SECRET', regex: /\b(?:GDPR|HIPAA|CCPA|SOC2|ISO27001)[-_: ]?AUDIT[-_: ]?\d{4}\b/gi },
|
|
@@ -214,10 +243,10 @@ let PROFILE_RULES = {
|
|
|
214
243
|
{ type: 'ID', regex: /\b(?:SAR|DSAR)[-_\/: ]?[A-Z0-9-/]+\b/gi }
|
|
215
244
|
],
|
|
216
245
|
ccpa: [
|
|
217
|
-
{ type: 'ID', regex: /\
|
|
246
|
+
{ type: 'ID', regex: /\b(?:DL|DRIVER['’]?S?\s+LICENSE)[:\s#-]*[A-Z0-9]{6,12}\b/gi },
|
|
218
247
|
{ type: 'LOCATION', regex: /\b-?\d{1,3}\.\d{4,6}[° ]?[NSns],\s*-?\d{1,3}\.\d{4,6}[° ]?[EWew]\b/g },
|
|
219
|
-
{ type: '
|
|
220
|
-
{ type: 'ID', regex: /\bACCOUNT[ -]?(?:ID|NUM|NUMBER)[:\s][A-Z0-9]{6,20}\b/gi }
|
|
248
|
+
{ type: 'ID', regex: /\b(?:CCPA|CPRA)[-_: ]?OPT[-_ ]OUT\b/gi },
|
|
249
|
+
{ type: 'ID', regex: /\bACCOUNT[ -]?(?:ID|NUM|NUMBER)[:\s]+[A-Z0-9]{6,20}\b/gi }
|
|
221
250
|
],
|
|
222
251
|
engineering: [
|
|
223
252
|
{ type: 'SECRET', regex: /(?<=\b(?:DB|POSTGRES|REDIS|MYSQL|AWS|SECRET|PASSWORD|TOKEN|API|KEY)[A-Z0-9_]*\s*[:=]\s*["']?)[A-Za-z0-9_-]{10,}/gi },
|
|
@@ -226,7 +255,7 @@ let PROFILE_RULES = {
|
|
|
226
255
|
{ type: 'ID', regex: /\b[a-z0-9](?:[-a-z0-9]*[a-z0-9])?\.svc\.cluster\.local\b/g }
|
|
227
256
|
],
|
|
228
257
|
agents: [
|
|
229
|
-
{ type: 'ID', regex: /\b(?:AGENT|VECTOR|EMBEDDING)[-_: ]?[A-Z0-9]{8,}\b/gi },
|
|
258
|
+
{ type: 'ID', regex: /\b(?:AGENT|VECTOR|EMBEDDING)[-_: ]?(?:ID[-_: ]?)?[A-Z0-9]{8,}\b/gi },
|
|
230
259
|
{ type: 'ID', regex: /\bTASK[-_: ]?[A-Z0-9]{5,15}\b/gi },
|
|
231
260
|
{ type: 'SECRET', regex: /\b(?:SYS_PROMPT|SYSTEM_PROMPT|OPENAI_API_KEY)[-_: ]?[A-Za-z0-9_-]{10,}\b/gi }
|
|
232
261
|
],
|
|
@@ -247,7 +276,7 @@ let PROFILE_RULES = {
|
|
|
247
276
|
{ type: 'SECRET', regex: /\b(?:CONFIG|KUBECONFIG|TFSTATE)[-_: ]?[A-Z0-9]{6,15}\b/gi }
|
|
248
277
|
],
|
|
249
278
|
personal: [
|
|
250
|
-
{ type: '
|
|
279
|
+
{ type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
|
|
251
280
|
{ type: 'SECRET', regex: /\b(?:PASSWORD|PWD|SECRET|PIN)[\s:]*[\S]{4,20}\b/gi },
|
|
252
281
|
{ type: 'PHONE', regex: /\b(?:WIFE|HUSBAND|PARTNER|MOM|DAD)[\s:]+(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}\b/gi }
|
|
253
282
|
],
|
|
@@ -294,10 +323,46 @@ let PROFILE_RULES = {
|
|
|
294
323
|
{ type: 'ID', regex: /\b(?:Batch|Lot|Serial)\s*(?:No[.]?|Number|#)?[:\s#]*[A-Z0-9][A-Z0-9\-]{3,14}\b/gi },
|
|
295
324
|
{ type: 'PHI', regex: /\b(?:Dose|Dosage)[:\s]+\d+(?:\.\d+)?\s*(?:mg|mcg|mL|IU|units?)\b/gi },
|
|
296
325
|
{ type: 'ID', regex: /\b(?:CRF|eCRF|Case\s+Report\s+Form)\s*(?:No|Page|ID)?[:\s#]*[A-Z0-9]{2,10}\b/gi }
|
|
326
|
+
],
|
|
327
|
+
underwriting: [
|
|
328
|
+
// Employer Corporate / Business Names
|
|
329
|
+
{ type: 'NAME', isContextName: true, regex: /\b(?:[A-Z][A-Za-z0-9&.,'-]*[ \t\xA0]+){1,5}(?:Inc\.?|LLC|Corp\.?|Corporation|Ltd\.?|Limited|Co\.?|Company|Group|Holdings|Solutions|Services|Technologies|Logistics|Industries|Capital|Bank|Partners|LLP|PLLC)(?:\s+(?:LLC|Inc\.?|Corp\.?|Ltd\.?|USA|Group))?\b/g },
|
|
330
|
+
{ type: 'NAME', isContextName: true, regex: /(?:Employer|Company|Organization|Business)\s*(?:Name)?[\s:#]+([A-Za-z0-9&.,\s'-]{2,40}?)(?=\r?\n|$|\s{3,}|\t|Address|EIN|FEIN|Phone|W-2|Rate|Pay|Wage)/gi },
|
|
331
|
+
|
|
332
|
+
// W-2 & Tax Identifiers
|
|
333
|
+
{ type: 'ID', regex: /(?:\b(?:Box\s+d\b|d\.\s*(?:Control|#)?|d\s+Control)\s*(?:number|no\.?|#|num)?[:\s#]*|\bControl\s*(?:number|no\.?|#|num)[:\s#]*|\bControl[:#]\s*)([A-Za-z0-9-]{3,30})/gi },
|
|
334
|
+
{ type: 'ID', regex: /\b(?:EIN|FEIN|Tax\s+ID)[:\s#]*\d{2}-\d{7}\b/gi },
|
|
335
|
+
{ type: 'ID', regex: /\b\d{2}-\d{7}\b/g },
|
|
336
|
+
{ type: 'ID', regex: /\b\d{3}-\d{2}-\d{4}\b/g },
|
|
337
|
+
{ type: 'ID', regex: /\b(?:XXX|xxx|\*\*\*)[ -]?(?:XX|xx|\*\*)[ -]?\d{4}\b/g },
|
|
338
|
+
|
|
339
|
+
// Employee, Loan & Payroll IDs
|
|
340
|
+
{ type: 'ID', regex: /\b(?:Employee|Emp|EE|Worker|Borrower|Badge|Advisor|Producer|Agent|Applicant|File)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Z0-9-]{3,20}\b/gi },
|
|
341
|
+
{ type: 'ID', regex: /\b(?:Pay\s+Group|Cost\s+Center|Dept|Department)[:\s#]*[A-Za-z0-9_-]{2,30}\b/gi },
|
|
342
|
+
{ type: 'ID', regex: /\b(?:Loan|Application|Deal|Borrower|File)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Za-z0-9-]{4,25}\b/gi },
|
|
343
|
+
|
|
344
|
+
// Direct Deposit, Bank Accounts & Routing Numbers (Masked & Unmasked)
|
|
345
|
+
{ type: 'FINANCIAL', regex: /\b(?:Account|Acct|Checking|Savings|Direct\s+Deposit)\s*(?:#|ID|No\.?|Number)?[:\s#]*(?:[\*xX•.-]{3,}\d{2,6}|\d{4}[-\s]?\d{4}[-\s]?\d{2,6})\b/gi },
|
|
346
|
+
{ type: 'FINANCIAL', regex: /\b(?:ABA|Routing|RTN)\s*(?:#|ID|No\.?|Number)?[:\s#]*\d{9}\b/gi },
|
|
347
|
+
|
|
348
|
+
// Borrower & Co-Borrower Names (ALL-CAPS, Payroll, Title Case)
|
|
349
|
+
{ type: 'NAME', regex: /\b[A-Z]{2,25},\s+[A-Z]{2,25}(?:\s+[A-Z]\.?|\s+[A-Z]{2,25})*\b/g },
|
|
350
|
+
{ type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+(?:\p{Lu}\.?|[A-Z][a-z]+))?[ \t\xA0]+\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
351
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:Borrower|Co-Borrower|Applicant|Co-Applicant|Employee|Worker|Taxpayer|Candidate|Primary\s+Borrower|Joint\s+Borrower|Account\s+Holder|Insured|Client)\s*(?:Name)?|First\s+name|Given\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9&.,\s'-]{2,35}?)(?=\r?\n|$|\s{3,}|\t|SSN|EIN|DOB|Address|Phone|Rate|Pay|Wage|Date|Box|Last|Surname)/gi },
|
|
352
|
+
{ type: 'NAME', isContextName: true, regex: /(?:Last\s+name|Surname|Family\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z'-]{2,30})/gi },
|
|
353
|
+
|
|
354
|
+
// Addresses & Locations
|
|
355
|
+
{ type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:Borrower(?:'s)?|Co-Borrower(?:'s)?|Employee(?:'s)?|Employer(?:'s)?|Home|Mailing|Property|Physical)\s+address|(?:(?:\bBox\s+f\b|\bf\.\s*|\bf\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?address))[\s:#]+([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|City|State|ZIP|SSN|EIN|Phone|Box|\d+\b)/gi },
|
|
356
|
+
{ type: 'ADDRESS', regex: /\b\d{1,6}[ \t\xA0]+(?:[A-Za-z0-9.-]+[ \t\xA0]+){1,4}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Hwy|Highway|Cir|Circle|Trl|Trail|Loop|Row|Pike|Box|PO Box|P\.O\.[ \t\xA0]*Box)\b(?:[ \t\xA0]*,?[ \t\xA0]*(?:Apt|Apartment|Suite|Ste|Unit|#|Fl|Floor|Bldg|Building)\.?[ \t\xA0]*[A-Za-z0-9-]+)?/gi },
|
|
357
|
+
{ type: 'ADDRESS', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
|
|
358
|
+
{ type: 'ADDRESS', regex: /\b\d{5}-\d{4}\b/g },
|
|
359
|
+
{ type: 'LOCATION', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\b/g }
|
|
297
360
|
]
|
|
298
361
|
};
|
|
299
362
|
|
|
300
363
|
let NAME_STOP_LIST = new Set([
|
|
364
|
+
'clinical note', 'case note', 'prod log', 'siem alert', 'hr review', 'crm export', 'bank statement', 'file export', 'database row', 'lease application', 'strategy export', 'action log', 'glossary query', 'tool comparison', 'call transcript', 'zendesk ticket', 'board minutes', 'agent context', 'config dump', 'database dump', 'patient note', 'medical record', 'admission note', 'discharge summary', 'progress note', 'hiring review', 'security audit', 'incident response', 'server log', 'system log', 'api response', 'error log', 'audit log', 'debug log',
|
|
365
|
+
'tax statement', 'wage and tax statement', 'wage and tax', 'wage statement', 'earning statement', 'earnings statement', 'pay statement', 'pay stub', 'paystub', 'withholding statement',
|
|
301
366
|
'case no', 'account no', 'client no', 'ref no', 'matter no',
|
|
302
367
|
'affected user', 'incident date', 'incident type', 'incident report',
|
|
303
368
|
'review period', 'review date', 'salary band', 'salary range',
|
|
@@ -339,7 +404,9 @@ let NAME_STOP_LIST = new Set([
|
|
|
339
404
|
'step 1', 'step 2', 'step 3', 'step 4', 'step 5',
|
|
340
405
|
'page 1', 'page 2', 'page 3', 'page 4', 'page 5',
|
|
341
406
|
'cs101', 'course cs101',
|
|
342
|
-
// Expanded Stop List (Common nouns
|
|
407
|
+
// Expanded Stop List (Common nouns, command phrases, legal, prompt, chess, and animation terms)
|
|
408
|
+
'docket number', 'docket numbers', 'dockets section', 'case name', 'case names', 'case number', 'case numbers', 'law firm', 'law firms', 'counsel stack', 'counselstack', 'counselstack connector', 'tier 0', 'tier 1', 'tier 2', 'tier 3', 'tier 4', 'do not', 'do not write', 'specific permission', 'write again', 'without permission', 'without specific permission', 'on screen', 'in report', 'own line', 'connector access', 'prompt instruction', 'prompt instructions', 'finding report', 'findings report',
|
|
409
|
+
'white bishop', 'black bishop', 'white knight', 'black knight', 'white king', 'black king', 'white queen', 'black queen', 'white rook', 'black rook', 'white pawn', 'black pawn', 'chess piece', 'chess pieces', 'chess game', 'chess match', 'disney-pixar', 'disney pixar', 'pixar animation', 'close-up', 'close up',
|
|
343
410
|
'january', 'february', 'march', 'april', 'may', 'june', 'july', 'august', 'september', 'october', 'november', 'december',
|
|
344
411
|
'monday', 'tuesday', 'wednesday', 'thursday', 'friday', 'saturday', 'sunday',
|
|
345
412
|
'yesterday', 'tomorrow', 'today', 'last week', 'next month', 'early morning', 'late night',
|
|
@@ -373,24 +440,104 @@ let NAME_STOP_LIST = new Set([
|
|
|
373
440
|
'class name', 'function name', 'variable name', 'database table', 'schema name', 'index name', 'query result', 'error message', 'warning message', 'log entry', 'debug log', 'stack trace',
|
|
374
441
|
'staff member', 'team member', 'board member', 'board meeting', 'committee member', 'executive board',
|
|
375
442
|
'email us', 'contact us', 'about us', 'sign in', 'sign out',
|
|
443
|
+
'driver license', 'drivers license', 'opt out', 'opt-out', 'ccpa opt', 'cpra opt', 'spoiler unreleased', 'unreleased draft',
|
|
376
444
|
'lighting', 'keyboard', 'creating', 'building', 'training', 'planning', 'starting', 'painting', 'printing', 'returned', 'released', 'required', 'accepted', 'imported', 'services', 'products', 'accounts', 'settings', 'partners', 'keywords', 'keystone', 'keyspace', 'keynotes', 'keychain'
|
|
377
|
-
, 'quarterly results', 'strategic planning', 'market research', 'customer base', 'privacy settings', 'account settings', 'security settings', 'download now', 'free trial', 'limited time', 'copyright protected', 'all rights', 'rights reserved', 'credit score', 'monthly rent', 'lease application', 'property address', 'reference number', 'additional identifier', 'lease agreement', 'hiring review', 'candidate name', 'privacy policy', 'terms of service', 'machine learning', 'artificial intelligence', 'generative ai', 'silicon valley', 'google cloud', 'amazon web', 'data science', 'operating system', 'software engineer', 'product manager', 'project manager', 'data analyst', 'gross margin', 'revenue growth', 'source code', 'version control', 'large language model'
|
|
445
|
+
, 'quarterly results', 'strategic planning', 'market research', 'customer base', 'privacy settings', 'account settings', 'security settings', 'download now', 'free trial', 'limited time', 'copyright protected', 'all rights', 'rights reserved', 'credit score', 'monthly rent', 'lease application', 'property address', 'reference number', 'additional identifier', 'lease agreement', 'hiring review', 'candidate name', 'privacy policy', 'terms of service', 'machine learning', 'artificial intelligence', 'generative ai', 'silicon valley', 'google cloud', 'amazon web', 'data science', 'operating system', 'software engineer', 'product manager', 'project manager', 'data analyst', 'gross margin', 'revenue growth', 'source code', 'version control', 'large language model',
|
|
446
|
+
'wages', 'wage', 'tips', 'compensation', 'withheld', 'withholding', 'medicare', 'deductions', 'deduction',
|
|
447
|
+
'regular', 'hours', 'holiday', 'overtime', 'commission', 'bonus', 'bonuses', 'records', 'record', 'statement', 'statements', 'rate', 'rates', 'current', 'ytd', 'benefits', 'taxable', 'pre-tax', 'post-tax', 'reimbursements', 'reimbursement', 'fica', 'oasdi', 'disability', 'unemployment', 'sui', 'sdi', 'std', 'ltd', 'exemptions', 'exemption', 'allowances', 'allowance', 'filing', 'status', 'single', 'married', 'head', 'household', 'advice', 'frequency', 'bi-weekly', 'biweekly', 'weekly', 'monthly', 'semi-monthly', 'direct', 'deposit', 'routing', 'box', 'boxes', 'code', 'control', 'omb', 'copy', 'instructions', 'information', 'deferred', 'adoption', 'statutory', 'third-party', 'sick', 'form', 'schedule', 'w-2', 'w2', 'w-4', 'w4', '1099', 'k-1', '1040', 'fed', 'med', 'fwt', 'swt', 'fed w/h', 'fed med', 'locality', 'state wages', 'state tax', 'local wages', 'local tax', 'allocated', 'nonqualified', 'suff', 'suffix', 'allocated tips', 'advance eic', 'advance eic payment', 'dependent care', 'dependent care benefits', 'nonqualified plans', 'statutory employee', 'retirement plan', 'third-party sick pay']);
|
|
378
448
|
|
|
379
|
-
let JARGON_WORDS = new Set(['step', 'page', 'grade', 'version', 'course', 'class', 'follow', 'chapter', 'lesson', 'unit', 'marketing', 'manager', 'specialist', 'science', 'administration', 'university', 'skills', 'leadership', 'communication', 'working', 'proficiency', 'decision', 'driven', 'experience', 'summary', 'bachelor', 'ads', 'solutions', 'positioning', 'acquisition', 'strategy', 'research', 'database', 'forecast', 'interest', 'prepared', 'merchant', 'document', 'feedback', 'template', 'campaign', 'partners', 'settings', 'keystone', 'llm', 'gpt', 'chatgpt', 'openai', 'anthropic', 'claude', 'gemini', 'api', 'json', 'xml', 'html', 'css', 'javascript', 'python', 'policy', 'terms', 'conditions', 'release', 'sprint', 'deployment', 'cluster', 'instance', 'package', 'module', 'revenue', 'margin', 'gross', 'quarter', 'system', 'code', 'data', 'cloud', 'server', 'database', 'artificial', 'intelligence', 'learning', 'generative']);
|
|
449
|
+
let JARGON_WORDS = new Set(['step', 'page', 'grade', 'version', 'course', 'class', 'follow', 'chapter', 'lesson', 'unit', 'marketing', 'manager', 'specialist', 'science', 'administration', 'university', 'skills', 'leadership', 'communication', 'working', 'proficiency', 'decision', 'driven', 'experience', 'summary', 'bachelor', 'ads', 'solutions', 'positioning', 'acquisition', 'strategy', 'research', 'database', 'forecast', 'interest', 'prepared', 'merchant', 'document', 'feedback', 'template', 'campaign', 'partners', 'settings', 'keystone', 'llm', 'gpt', 'chatgpt', 'openai', 'anthropic', 'claude', 'gemini', 'api', 'json', 'xml', 'html', 'css', 'javascript', 'python', 'golang', 'typescript', 'rust', 'fastapi', 'snowflake', 'kubernetes', 'terraform', 'docker', 'redis', 'kafka', 'pytorch', 'policy', 'terms', 'conditions', 'release', 'sprint', 'deployment', 'cluster', 'instance', 'package', 'module', 'revenue', 'margin', 'gross', 'quarter', 'system', 'code', 'data', 'cloud', 'server', 'database', 'artificial', 'intelligence', 'learning', 'generative', 'regular', 'hours', 'holiday', 'earnings', 'deductions', 'withheld', 'withholding', 'taxes', 'medicare', 'benefits', 'reimbursements', 'compensation', 'wages', 'code']);
|
|
380
450
|
|
|
381
451
|
let NOT_NAME_WORDS = new Set([
|
|
382
452
|
// Grammatical & Sentence Starters
|
|
383
|
-
'the', 'a', 'an', 'this', 'that', 'these', 'those', 'my', 'your', 'his', 'her', 'their', 'our', 'its', 'it', 'he', 'she', 'they', 'we', 'i', 'you', 'who', 'whom', 'which', 'what', 'whose', 'why', 'how', 'when', 'where', 'with', 'for', 'from', 'by', 'to', 'at', 'in', 'on', 'of', 'about', 'as', 'into', 'through', 'during', 'before', 'after', 'above', 'below', 'and', 'but', 'or', 'so', 'yet',
|
|
453
|
+
'the', 'a', 'an', 'this', 'that', 'these', 'those', 'my', 'your', 'his', 'her', 'their', 'our', 'its', 'it', 'he', 'she', 'they', 'we', 'i', 'you', 'who', 'whom', 'which', 'what', 'whose', 'why', 'how', 'when', 'where', 'with', 'for', 'from', 'by', 'to', 'at', 'in', 'on', 'of', 'about', 'as', 'into', 'through', 'during', 'before', 'after', 'above', 'below', 'and', 'but', 'or', 'so', 'yet', 'im', "i'm", "you're", "they're", "we're", "it's", "he's", "she's", "that's", "there's", "what's", "who's", "i've", "you've", "we've", "they've", "i'll", "you'll", "we'll", "they'll", "i'd", "you'd", "we'd", "they'd",
|
|
454
|
+
// Verbs, Auxiliaries, Commands & Imperatives
|
|
455
|
+
'do', 'does', 'did', 'done', 'doing', 'dont', "don't", 'doesnt', "doesn't", 'didnt', "didn't", 'not', 'no', 'never', 'always',
|
|
456
|
+
'be', 'is', 'am', 'are', 'was', 'were', 'been', 'being',
|
|
457
|
+
'have', 'has', 'had', 'having',
|
|
458
|
+
'can', 'could', 'may', 'might', 'must', 'shall', 'should', 'will', 'would', 'wont', "won't", 'wouldnt', "wouldn't", 'shouldnt', "shouldn't", 'couldnt', "couldn't", 'cant', "can't", 'cannot',
|
|
459
|
+
'write', 'writing', 'written', 'writes', 'read', 'reading', 'reads',
|
|
460
|
+
'wait', 'waiting', 'waited', 'waits', 'place', 'placing', 'placed', 'places',
|
|
461
|
+
'display', 'displaying', 'displayed', 'displays',
|
|
462
|
+
'provide', 'providing', 'provided', 'provides',
|
|
463
|
+
'show', 'showing', 'shown', 'shows',
|
|
464
|
+
'tell', 'telling', 'told', 'tells',
|
|
465
|
+
'ask', 'asking', 'asked', 'asks',
|
|
466
|
+
'use', 'using', 'used', 'uses',
|
|
467
|
+
'select', 'selecting', 'selected', 'selects',
|
|
468
|
+
'find', 'finding', 'findings', 'found', 'finds',
|
|
469
|
+
'reference', 'referencing', 'referenced', 'references',
|
|
470
|
+
'access', 'accessing', 'accessed', 'accesses',
|
|
471
|
+
'note', 'noting', 'noted', 'notes',
|
|
472
|
+
'get', 'getting', 'got', 'gotten', 'gets',
|
|
473
|
+
'make', 'making', 'made', 'makes',
|
|
474
|
+
'give', 'giving', 'given', 'gives',
|
|
475
|
+
'take', 'taking', 'took', 'taken', 'takes',
|
|
476
|
+
'put', 'putting', 'puts',
|
|
477
|
+
'set', 'setting', 'sets',
|
|
478
|
+
'keep', 'keeping', 'kept', 'keeps',
|
|
479
|
+
'let', 'letting', 'lets',
|
|
480
|
+
'leave', 'leaving', 'left', 'leaves',
|
|
481
|
+
'run', 'running', 'ran', 'runs',
|
|
482
|
+
'stop', 'stopping', 'stopped', 'stops',
|
|
483
|
+
'start', 'starting', 'started', 'starts',
|
|
484
|
+
'check', 'checking', 'checked', 'checks',
|
|
485
|
+
'print', 'printing', 'printed', 'prints',
|
|
486
|
+
'generate', 'generating', 'generated', 'generates',
|
|
487
|
+
'create', 'creating', 'created', 'creates',
|
|
488
|
+
'build', 'building', 'built', 'builds',
|
|
489
|
+
'include', 'including', 'included', 'includes',
|
|
490
|
+
'exclude', 'excluding', 'excluded', 'excludes',
|
|
491
|
+
'format', 'formatting', 'formatted', 'formats',
|
|
492
|
+
'change', 'changing', 'changed', 'changes',
|
|
493
|
+
'send', 'sending', 'sent', 'sends',
|
|
494
|
+
'receive', 'receiving', 'received', 'receives',
|
|
495
|
+
'delete', 'deleting', 'deleted', 'deletes',
|
|
496
|
+
'remove', 'removing', 'removed', 'removes',
|
|
497
|
+
'insert', 'inserting', 'inserted', 'inserts',
|
|
498
|
+
'update', 'updating', 'updated', 'updates',
|
|
499
|
+
'review', 'reviewing', 'reviewed', 'reviews',
|
|
500
|
+
'allow', 'allowing', 'allowed', 'allows',
|
|
501
|
+
'deny', 'denying', 'denied', 'denies',
|
|
502
|
+
'require', 'requiring', 'required', 'requires',
|
|
503
|
+
'turn', 'turning', 'turned', 'turns',
|
|
504
|
+
'switch', 'switching', 'switched', 'switches',
|
|
505
|
+
'enable', 'enabling', 'enabled', 'enables',
|
|
506
|
+
'disable', 'disabling', 'disabled', 'disables',
|
|
507
|
+
'ensure', 'ensuring', 'ensured', 'ensures',
|
|
508
|
+
'verify', 'verifying', 'verified', 'verifies',
|
|
509
|
+
'execute', 'executing', 'executed', 'executes',
|
|
510
|
+
'test', 'testing', 'tested', 'tests',
|
|
511
|
+
'install', 'installing', 'installed', 'installs',
|
|
512
|
+
'uninstall', 'uninstalling', 'uninstalled', 'uninstalls',
|
|
513
|
+
'suppose', 'supposed', 'supposing', 'supposes',
|
|
514
|
+
'respond', 'responding', 'responded', 'responds',
|
|
515
|
+
'preserve', 'preserving', 'preserved', 'preserves',
|
|
516
|
+
'replace', 'replacing', 'replaced', 'replaces',
|
|
517
|
+
// Adverbs, Prepositions, Conjunctions & Modifiers
|
|
518
|
+
'again', 'without', 'with', 'within', 'specific', 'specifically', 'permission', 'permissions',
|
|
519
|
+
'underneath', 'above', 'below', 'between', 'among', 'together', 'separately', 'instead',
|
|
520
|
+
'also', 'too', 'either', 'neither', 'both', 'each', 'every', 'all', 'some', 'any', 'none',
|
|
521
|
+
'only', 'just', 'already', 'currently', 'more', 'most', 'less', 'least',
|
|
522
|
+
'very', 'quite', 'rather', 'such', 'same', 'different', 'other', 'others', 'another',
|
|
523
|
+
'like', 'unlike', 'similar', 'complete', 'completely', 'entire', 'entirely',
|
|
524
|
+
'exact', 'exactly', 'approximate', 'approximately', 'general', 'generally',
|
|
525
|
+
'direct', 'directly', 'indirect', 'indirectly', 'total', 'totally', 'full', 'fully',
|
|
526
|
+
'partial', 'partially', 'own', 'proper', 'properly',
|
|
527
|
+
'now', 'then', 'soon', 'later', 'here', 'there', 'everywhere', 'nowhere', 'somewhere', 'anywhere',
|
|
528
|
+
'inside', 'outside', 'before', 'after', 'since', 'until', 'till',
|
|
529
|
+
'while', 'whereas', 'unless', 'although', 'though', 'even', 'because',
|
|
530
|
+
'therefore', 'however', 'furthermore', 'moreover', 'meanwhile', 'otherwise', 'besides', 'further',
|
|
384
531
|
// Greetings & Salutations
|
|
385
532
|
'hello', 'hi', 'hey', 'dear', 'greetings',
|
|
386
533
|
// Document & Resume Structure
|
|
387
|
-
'summary', 'experience', 'education', 'skills', 'languages', 'project', 'history', 'background', 'objective', 'profile', 'awards', 'honors', 'certifications', 'publications', 'interests', 'references',
|
|
534
|
+
'summary', 'experience', 'education', 'skills', 'languages', 'project', 'history', 'background', 'objective', 'profile', 'awards', 'honors', 'certifications', 'publications', 'interests', 'references', 'statement', 'statements', 'form', 'forms',
|
|
388
535
|
// Business & Job Roles
|
|
389
536
|
'manager', 'director', 'specialist', 'analyst', 'engineer', 'developer', 'consultant', 'officer', 'representative', 'agent', 'lead', 'leader', 'president', 'coordinator', 'admin', 'administrator', 'executive', 'founder', 'partner', 'intern', 'trainee', 'advisor', 'head', 'vp', 'chief',
|
|
390
537
|
// Departments & Fields
|
|
391
538
|
'marketing', 'sales', 'engineering', 'finance', 'accounting', 'legal', 'operations', 'support', 'recruiting', 'talent', 'acquisition', 'compliance', 'security', 'technical', 'development', 'product', 'design', 'creative', 'strategy', 'planning', 'analytics', 'science', 'business', 'administration',
|
|
392
539
|
// Tools & Tech Concepts
|
|
393
|
-
'google', 'ads', 'analytics', 'meta', 'hubspot', 'crm', 'salesforce', 'wordpress', 'mailchimp', 'adobe', 'figma', 'canva', 'slack', 'zoom', 'teams', 'microsoft', 'office', 'excel', 'word', 'powerpoint', 'notion', 'jira', 'confluence', 'github', 'gitlab', 'aws', 'azure', 'cloud', 'database', 'sql', 'python', 'java', 'javascript', 'html', 'css', 'react', 'node', 'api', 'saas', 'b2b', 'b2c', 'url', 'domain', 'website', 'app', 'application', 'software', 'email', 'phone', 'contact', 'address',
|
|
540
|
+
'google', 'ads', 'analytics', 'meta', 'hubspot', 'crm', 'salesforce', 'wordpress', 'mailchimp', 'adobe', 'figma', 'canva', 'slack', 'zoom', 'teams', 'microsoft', 'office', 'excel', 'word', 'powerpoint', 'notion', 'jira', 'confluence', 'github', 'gitlab', 'aws', 'gcp', 'azure', 'cloud', 'database', 'sql', 'python', 'golang', 'typescript', 'rust', 'fastapi', 'snowflake', 'kubernetes', 'terraform', 'docker', 'redis', 'kafka', 'pytorch', 'java', 'javascript', 'html', 'css', 'react', 'node', 'api', 'saas', 'b2b', 'b2c', 'url', 'domain', 'website', 'app', 'application', 'software', 'email', 'phone', 'contact', 'address',
|
|
394
541
|
|
|
395
542
|
// General Academic & Professional vocabulary
|
|
396
543
|
'bachelor', 'master', 'doctor', 'associate', 'degree', 'university', 'college', 'school', 'institute', 'academy', 'graduated', 'major', 'minor', 'gpa', 'cum', 'laude', 'honors', 'deans', 'list', 'scholarship',
|
|
@@ -399,15 +546,29 @@ let NOT_NAME_WORDS = new Set([
|
|
|
399
546
|
// Common Resume / Business Phrases
|
|
400
547
|
'results-driven', 'data-driven', 'customer-centric', 'detail-oriented', 'cross-functional', 'self-motivated', 'time-management', 'problem-solving', 'fast-paced', 'year-over-year',
|
|
401
548
|
// Legal & Trust terms
|
|
402
|
-
'trust', 'trustee', 'co-trustee', 'settlor', 'grantor', 'beneficiary', 'agreement', 'will', 'estate', 'witness', 'declaration', 'signatory', 'testator', 'notary', 'commission', 'county', 'state', 'court', 'article', 'section', 'paragraph', 'schedule', 'exhibit', 'amendment', 'addendum', 'power', 'attorney', 'guardian', 'executor', 'administrator', 'survivor', 'predecessor', 'successor', 'whereof', 'hereby', 'thereby', 'herein', 'therein', 'witnesseth', 'whereas', 'therefore', 'now', 'dated', 'effective',
|
|
549
|
+
'trust', 'trustee', 'co-trustee', 'settlor', 'grantor', 'beneficiary', 'agreement', 'will', 'estate', 'witness', 'declaration', 'signatory', 'testator', 'notary', 'commission', 'county', 'state', 'court', 'article', 'section', 'paragraph', 'schedule', 'exhibit', 'amendment', 'addendum', 'power', 'attorney', 'guardian', 'executor', 'administrator', 'survivor', 'predecessor', 'successor', 'whereof', 'hereby', 'thereby', 'herein', 'therein', 'witnesseth', 'whereas', 'therefore', 'now', 'dated', 'effective', 'matter', 'case', 'cases', 'docket', 'dockets', 'number', 'numbers', 'firm', 'firms', 'lawyer', 'lawyers', 'counsel', 'counsels', 'counselstack', 'tier', 'tiers', 'finding', 'findings', 'connector', 'connectors', 'platform', 'platforms',
|
|
550
|
+
// Medical & Clinical terms
|
|
551
|
+
'clinical', 'note', 'notes', 'dx', 'rx', 'tx', 'hx', 'px', 'sx', 'type', 'diabetes', 'referred', 'referral', 'diagnosed', 'diagnosis', 'patient', 'insurance', 'bcbs', 'mrn', 'dob',
|
|
552
|
+
// Tax & Payroll terms
|
|
553
|
+
'wages', 'wage', 'tips', 'compensation', 'withheld', 'withholding', 'medicare', 'deductions', 'deduction', 'earning', 'earnings', 'gross', 'net', 'pay', 'payroll', 'paystub', 'taxable', 'exempt', 'allowance', 'allowances', 'regular', 'hours', 'holiday', 'overtime', 'commission', 'bonus', 'bonuses', 'records', 'record', 'statement', 'statements', 'rate', 'rates', 'current', 'ytd', 'benefits', 'taxable', 'pre-tax', 'post-tax', 'reimbursements', 'reimbursement', 'fica', 'oasdi', 'disability', 'unemployment', 'sui', 'sdi', 'std', 'ltd', 'exemptions', 'exemption', 'allowances', 'allowance', 'filing', 'status', 'single', 'married', 'head', 'household', 'advice', 'frequency', 'bi-weekly', 'biweekly', 'weekly', 'monthly', 'semi-monthly', 'direct', 'deposit', 'routing', 'box', 'boxes', 'code', 'control', 'omb', 'copy', 'instructions', 'information', 'deferred', 'adoption', 'statutory', 'third-party', 'sick', 'form', 'schedule', 'w-2', 'w2', 'w-4', 'w4', '1099', 'k-1', '1040', 'fed', 'med', 'fwt', 'swt', 'fed w/h', 'fed med', 'locality', 'state wages', 'state tax', 'local wages', 'local tax', 'allocated', 'nonqualified',
|
|
554
|
+
// Common Web, UI, Compliance, Document & AI Terms (Suppresses false-positive Name detection on headlines, buttons, and badges)
|
|
555
|
+
'types', 'type', 'leaked', 'leak', 'leaks', 'masked', 'mask', 'masking', 'leave', 'screen', 'screens', 'risk', 'risks', 'cluster', 'clusters', 'parameter', 'parameters', 'processing', 'process', 'processed', 'verified', 'verify', 'verification', 'playground', 'guide', 'guides', 'protection', 'protect', 'corporate', 'enterprise', 'log', 'logs', 'airplane', 'mode', 'zero', 'trust', 'top', 'data', 'live', 'scrubber', 'scrub', 'scrubbed', 'note', 'notes', 'secret', 'secrets', 'card', 'cards', 'raw', 'input', 'output', 'contains', 'contain', 'contained', 'platform', 'solutions', 'pricing', 'company', 'news', 'dashboard', 'add', 'chrome', 'sample', 'samples', 'try', 'terms', 'privacy', 'policy', 'policies', 'home', 'compliance', 'framework', 'frameworks', 'audit', 'audits', 'receipt', 'receipts', 'overview', 'explore', 'vectors', 'vector', 'standard', 'standards', 'status', 'preview', 'view', 'actions', 'action', 'button', 'buttons', 'option', 'options', 'general', 'specialized', 'custom', 'rule', 'rules', 'token', 'tokens', 'value', 'values', 'session', 'sessions', 'local', 'server', 'servers', 'cloud', 'ram', 'memory', 'offline', 'online', 'client', 'browser', 'extension', 'workspace', 'workplace', 'pan', 'phi', 'pii', 'soc', 'soc2', 'gdpr', 'hipaa', 'ccpa', 'iso27001', 'pci', 'dss', 'nist', 'chatgpt', 'claude', 'gemini', 'copilot', 'perplexity', 'deepseek', 'qwen', 'grok', 'llama', 'mistral', 'ai', 'llm', 'prompt', 'prompts', 'transmission', 'transit', 'egress', 'neutralized', 'stripped', 'isolated', 'isolation', 'unlocked', 'locked', 'unlock', 'download', 'copy', 'dismiss', 'close', 'save', 'settings', 'protect', 'reveal', 'unmask', 'restore', 'restored', 'export', 'import',
|
|
556
|
+
// Games, Chess, and Playing Pieces
|
|
557
|
+
'bishop', 'bishops', 'knight', 'knights', 'rook', 'rooks', 'pawn', 'pawns', 'king', 'kings', 'queen', 'queens', 'chessboard', 'checkmate', 'stalemate', 'castling', 'en passant', 'chess',
|
|
558
|
+
// Colors & Visual Descriptors
|
|
559
|
+
'white', 'black', 'red', 'blue', 'green', 'yellow', 'orange', 'purple', 'pink', 'brown', 'gray', 'grey', 'dark', 'light', 'gold', 'silver', 'bronze',
|
|
560
|
+
// Animation, 3D Rendering & Prompt Terminology
|
|
561
|
+
'pixar', 'disney', 'animation', 'render', 'rendering', 'composition', 'cinematic', 'smooth', 'glides', 'glide', 'gliding', 'capture', 'captures', 'capturing', 'camera', 'orbit', 'orbits', 'orbiting', 'trapped', 'trap', 'trapping', 'square', 'squares', 'character', 'characters', 'expressive', 'living', 'texture', 'textures', 'reflection', 'reflections', 'grain', 'candlelight', 'wooden', 'polished', 'vertical', 'horizontal', 'macro', 'closeup', 'close-up', 'scene', 'scenes', 'shot', 'shots', 'shadow', 'shadows',
|
|
562
|
+
// Email, Outreach, Guest Posting & Agency Business Vocabulary
|
|
563
|
+
'guest', 'post', 'posts', 'posting', 'attached', 'attach', 'attachment', 'attachments', 'updated', 'update', 'updates', 'list', 'lists', 'line', 'lines', 'rate', 'rates', 'affordable', 'services', 'service', 'infotech', 'technologies', 'technology', 'agency', 'agencies', 'digital', 'marketing', 'traffic', 'smart', 'design', 'seo', 'per', 'host', 'hosting', 'sites', 'site', 'inbox', 'starred', 'snoozed', 'important', 'sent', 'drafts', 'draft', 'spam', 'bin', 'trash', 'purchases', 'travel', 'social', 'forums', 'promotions', 'promotion', 'reply', 'forward', 'labels', 'label', 'compose', 'message', 'messages', 'mailer', 'outreach', 'backlink', 'backlinks', 'domain', 'authority', 'da', 'dr', 'founder', 'ceo', 'cto', 'cfo', 'coo', 'vp', 'head', 'lead',
|
|
403
564
|
// US States
|
|
404
|
-
'california', 'texas', 'florida', 'york', 'illinois', 'pennsylvania', 'ohio', 'georgia', 'michigan', 'carolina', 'virginia', 'washington', 'arizona', 'massachusetts', 'tennessee', 'indiana', 'maryland', 'missouri', 'wisconsin', 'colorado', 'minnesota', 'alabama', 'louisiana', 'kentucky', 'oregon', 'oklahoma', 'connecticut', 'utah', 'iowa', 'nevada', 'arkansas', 'mississippi', 'kansas', 'new mexico', 'nebraska', 'idaho', 'hawaii', 'maine', 'new hampshire', 'rhode island', 'montana', 'delaware', 'south dakota', 'north dakota', 'alaska', 'vermont', 'wyoming'
|
|
405
|
-
|
|
565
|
+
'california', 'texas', 'florida', 'york', 'illinois', 'pennsylvania', 'ohio', 'georgia', 'michigan', 'carolina', 'virginia', 'washington', 'arizona', 'massachusetts', 'tennessee', 'indiana', 'maryland', 'missouri', 'wisconsin', 'colorado', 'minnesota', 'alabama', 'louisiana', 'kentucky', 'oregon', 'oklahoma', 'connecticut', 'utah', 'iowa', 'nevada', 'arkansas', 'mississippi', 'kansas', 'new mexico', 'nebraska', 'idaho', 'hawaii', 'maine', 'new hampshire', 'rhode island', 'montana', 'delaware', 'south dakota', 'north dakota', 'alaska', 'vermont', 'wyoming',
|
|
566
|
+
'f.3d', 'f.supp', 'u.s.c.', 'v.', 'plaintiff', 'defendant', 'v', 'u.s.', 'court', 'app.', 'reporter', 'cir.']);
|
|
406
567
|
|
|
407
568
|
const PROFILE_JARGON = {
|
|
408
|
-
medical: ['sleep', 'apnea', 'symptom', 'symptoms', 'trauma', 'hypertension', 'health', 'disease', 'condition', 'diagnosis', 'treatment', 'medication', 'dose', 'patient', 'clinic', 'surgery', 'therapy', 'alcohol', 'cannabis', 'blood', 'pressure', 'heart', 'rate', 'emergency', 'contact', 'relationship'],
|
|
569
|
+
medical: ['sleep', 'apnea', 'symptom', 'symptoms', 'trauma', 'hypertension', 'health', 'disease', 'condition', 'diagnosis', 'treatment', 'medication', 'dose', 'patient', 'clinic', 'surgery', 'therapy', 'alcohol', 'cannabis', 'blood', 'pressure', 'heart', 'rate', 'emergency', 'contact', 'relationship', 'type', 'diabetes', 'cancer', 'asthma', 'copd', 'covid', 'infection', 'syndrome', 'disorder', 'chronic', 'acute', 'illness', 'fever', 'allergy', 'pain', 'referral', 'referred', 'prescription', 'prescribed', 'doctor', 'physician', 'nurse', 'hospital', 'clinical', 'note', 'notes', 'dx', 'rx', 'tx', 'hx', 'px', 'sx', 'insurance', 'bcbs'],
|
|
409
570
|
realestate: ['escrow', 'tenant', 'landlord', 'lease', 'mortgage', 'appraisal', 'broker', 'property', 'zoning', 'parcel', 'rent', 'buyer', 'seller', 'agent', 'listing'],
|
|
410
|
-
legal: ['testator', 'notary', 'commission', 'county', 'court', 'affidavit', 'plaintiff', 'defendant', 'litigation', 'jurisdiction', 'agreement', 'contract', 'settlement', 'clause', 'article', 'section'],
|
|
571
|
+
legal: ['testator', 'notary', 'commission', 'county', 'court', 'affidavit', 'plaintiff', 'defendant', 'litigation', 'jurisdiction', 'agreement', 'contract', 'settlement', 'clause', 'article', 'section', 'matter', 'case'],
|
|
411
572
|
hr: ['candidate', 'employee', 'payroll', 'benefits', 'salary', 'vacation', 'supervisor', 'subordinate', 'performance', 'appraisal', 'interview', 'resume', 'applicant'],
|
|
412
573
|
sales: ['prospect', 'opportunity', 'quota', 'pipeline', 'deal', 'revenue', 'forecast', 'lead', 'churn', 'client', 'customer']
|
|
413
574
|
};
|
|
@@ -447,11 +608,14 @@ const PROFILE_JARGON = {
|
|
|
447
608
|
}
|
|
448
609
|
|
|
449
610
|
const PROFILE_ALIAS_MAP = {
|
|
611
|
+
'general': 'general',
|
|
612
|
+
'underwriting': 'underwriting', 'lending': 'underwriting', 'mortgage': 'underwriting', 'loan': 'underwriting', 'income': 'underwriting', 'income_verification': 'underwriting', 'payroll': 'underwriting', 'w2': 'underwriting', 'paystub': 'underwriting',
|
|
450
613
|
'medical': 'medical', 'healthcare': 'medical', 'health': 'medical', 'pharma': 'pharma',
|
|
451
|
-
'engineering': 'engineering', 'dev': 'engineering', 'tech': 'tech',
|
|
452
|
-
'finance': 'finance', 'bizops': 'bizops', 'sales': 'sales', 'wealthmgmt': 'wealthmgmt', 'insurance': 'insurance', 'accounting': 'accounting',
|
|
614
|
+
'engineering': 'engineering', 'dev': 'engineering', 'devops': 'engineering', 'tech': 'tech',
|
|
615
|
+
'finance': 'finance', 'bizops': 'bizops', 'sales': 'sales', 'wealthmgmt': 'wealthmgmt', 'wealth': 'wealthmgmt', 'insurance': 'insurance', 'accounting': 'accounting',
|
|
453
616
|
'legal': 'legal', 'compliance': 'compliance', 'ccpa': 'ccpa',
|
|
454
617
|
'hr': 'hr', 'security': 'security', 'marketing': 'marketing', 'support': 'support',
|
|
618
|
+
'realestate': 'realestate', 'academic': 'academic', 'agents': 'agents', 'ai_agents': 'agents', 'creative': 'creative', 'personal': 'personal'
|
|
455
619
|
};
|
|
456
620
|
|
|
457
621
|
function getActiveRules(activeProfile) {
|
|
@@ -459,7 +623,7 @@ const PROFILE_JARGON = {
|
|
|
459
623
|
if (activeProfile && activeProfile.toLowerCase() !== 'general') {
|
|
460
624
|
const canonicalProfile = PROFILE_ALIAS_MAP[activeProfile.toLowerCase()] || 'general';
|
|
461
625
|
if (canonicalProfile !== 'general' && PROFILE_RULES[canonicalProfile]) {
|
|
462
|
-
activeRules =
|
|
626
|
+
activeRules = PROFILE_RULES[canonicalProfile].concat(activeRules);
|
|
463
627
|
}
|
|
464
628
|
}
|
|
465
629
|
return activeRules;
|
|
@@ -482,14 +646,15 @@ const PROFILE_JARGON = {
|
|
|
482
646
|
|
|
483
647
|
if (customRules && customRules.length > 0) {
|
|
484
648
|
const sorted = [...customRules].sort((a, b) => {
|
|
485
|
-
const patternA = typeof a === 'string' ? a : a.pattern;
|
|
486
|
-
const patternB = typeof b === 'string' ? b : b.pattern;
|
|
649
|
+
const patternA = typeof a === 'string' ? a : (a.pattern || (a.regex ? a.regex.source : '') || '');
|
|
650
|
+
const patternB = typeof b === 'string' ? b : (b.pattern || (b.regex ? b.regex.source : '') || '');
|
|
487
651
|
return patternB.length - patternA.length;
|
|
488
652
|
});
|
|
489
653
|
|
|
490
654
|
sorted.forEach(cr => {
|
|
491
|
-
const pattern = typeof cr === 'string' ? cr : cr.pattern;
|
|
492
|
-
|
|
655
|
+
const pattern = typeof cr === 'string' ? cr : (cr.pattern || (cr.regex ? cr.regex.source : ''));
|
|
656
|
+
if (!pattern) return;
|
|
657
|
+
const label = typeof cr === 'string' ? 'CUSTOM' : (cr.label || cr.mask || cr.name || 'CUSTOM');
|
|
493
658
|
|
|
494
659
|
let rx;
|
|
495
660
|
try {
|
|
@@ -525,13 +690,23 @@ const PROFILE_JARGON = {
|
|
|
525
690
|
let matchedText = m[0];
|
|
526
691
|
let start = m.index;
|
|
527
692
|
|
|
528
|
-
if (
|
|
693
|
+
if (m.length > 1 && m[1] !== undefined && m[1] !== '') {
|
|
529
694
|
matchedText = m[1];
|
|
530
|
-
|
|
695
|
+
const relOffset = m[0].indexOf(m[1]);
|
|
696
|
+
if (relOffset !== -1) {
|
|
697
|
+
start = m.index + relOffset;
|
|
698
|
+
}
|
|
531
699
|
}
|
|
532
700
|
const end = start + matchedText.length;
|
|
533
701
|
|
|
534
702
|
if (rule.type !== 'NAME' && rule.type !== 'ADDRESS') {
|
|
703
|
+
const val = matchedText.toLowerCase().trim();
|
|
704
|
+
if (NAME_STOP_LIST.has(val) || NOT_NAME_WORDS.has(val)) {
|
|
705
|
+
// Skip if generic English dictionary term matched by greedy regex (e.g. SWIFT matching SCRUBBER or CONTAINS)
|
|
706
|
+
if (rule.type === 'FINANCIAL' || rule.type === 'ID' || rule.type === 'PRIVACY' || rule.type === 'SECRET') {
|
|
707
|
+
continue;
|
|
708
|
+
}
|
|
709
|
+
}
|
|
535
710
|
matches.push({ start, end, value: matchedText, type: rule.type });
|
|
536
711
|
} else {
|
|
537
712
|
let val = matchedText.toLowerCase().trim();
|
|
@@ -540,18 +715,26 @@ const PROFILE_JARGON = {
|
|
|
540
715
|
if (!rule.isContextName && rule.type === 'NAME') {
|
|
541
716
|
let words = val.split(/[ \t\xA0]+/);
|
|
542
717
|
let origWords = matchedText.split(/[ \t\xA0]+/);
|
|
543
|
-
while (words.length > 2 && (currentJargon.has(words[0]) || NOT_NAME_WORDS.has(words[0]))) {
|
|
718
|
+
while (words.length > 2 && (currentJargon.has(words[0]) || (words[0].length > 1 && NOT_NAME_WORDS.has(words[0])) || (words[0].replace(/[^\p{L}]/gu, '').length > 1 && NOT_NAME_WORDS.has(words[0].replace(/[^\p{L}]/gu, ''))))) {
|
|
544
719
|
origWords.shift();
|
|
545
720
|
words.shift();
|
|
546
721
|
const nextStart = matchedText.indexOf(origWords[0]);
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
722
|
+
if (nextStart !== -1) {
|
|
723
|
+
start += nextStart;
|
|
724
|
+
matchedText = matchedText.substring(nextStart);
|
|
725
|
+
val = matchedText.toLowerCase().trim();
|
|
726
|
+
} else {
|
|
727
|
+
break;
|
|
728
|
+
}
|
|
550
729
|
}
|
|
551
|
-
if (words.some(w =>
|
|
730
|
+
if (words.some(w => {
|
|
731
|
+
const cleanW = w.replace(/[^\p{L}]/gu, '');
|
|
732
|
+
if (cleanW.length <= 1) return false;
|
|
733
|
+
return currentJargon.has(w) || NOT_NAME_WORDS.has(w) || NOT_NAME_WORDS.has(cleanW) || NAME_STOP_LIST.has(w) || NAME_STOP_LIST.has(cleanW);
|
|
734
|
+
})) continue;
|
|
552
735
|
} else if (rule.type === 'ADDRESS') {
|
|
553
|
-
const
|
|
554
|
-
if (
|
|
736
|
+
const cleanVal = val.replace(/[.,;!?]/g, ' ').trim();
|
|
737
|
+
if (NAME_STOP_LIST.has(cleanVal) || currentJargon.has(cleanVal)) continue;
|
|
555
738
|
}
|
|
556
739
|
|
|
557
740
|
matches.push({ start, end: start + matchedText.length, value: matchedText, type: rule.type });
|
|
@@ -588,7 +771,7 @@ const PROFILE_JARGON = {
|
|
|
588
771
|
nameWords.forEach(w => {
|
|
589
772
|
if (w && w.length >= 2 && /^\p{Lu}/u.test(w)) {
|
|
590
773
|
const wl = w.toLowerCase();
|
|
591
|
-
if (!NOT_NAME_WORDS.has(wl) && !JARGON_WORDS.has(wl) && !NAME_STOP_LIST.has(wl)) {
|
|
774
|
+
if (!NOT_NAME_WORDS.has(wl) && !JARGON_WORDS.has(wl) && !NAME_STOP_LIST.has(wl) && !currentJargon.has(wl)) {
|
|
592
775
|
learnedNames.add(w);
|
|
593
776
|
}
|
|
594
777
|
}
|
|
@@ -641,13 +824,24 @@ const PROFILE_JARGON = {
|
|
|
641
824
|
return Array.from(aliases);
|
|
642
825
|
}
|
|
643
826
|
|
|
827
|
+
function formatToken(label, index, format = 'brackets') {
|
|
828
|
+
const cleanLabel = String(label || 'PII').replace(/[^A-Za-z0-9_]/g, '_').toUpperCase();
|
|
829
|
+
switch(format) {
|
|
830
|
+
case 'xml': return `<${cleanLabel}_${index}>`;
|
|
831
|
+
case 'mustache': return `{{${cleanLabel}_${index}}}`;
|
|
832
|
+
case 'underscores': return `__${cleanLabel}_${index}__`;
|
|
833
|
+
case 'brackets':
|
|
834
|
+
default: return `[${cleanLabel}_${index}]`;
|
|
835
|
+
}
|
|
836
|
+
}
|
|
837
|
+
|
|
644
838
|
function buildRestorationRegexAndRules(tokenMap) {
|
|
645
839
|
const ObjectKeys = Object.keys(tokenMap);
|
|
646
840
|
if (ObjectKeys.length === 0) return { compositeRegex: null, looseRules: [] };
|
|
647
841
|
|
|
648
842
|
const sortedKeys = [...ObjectKeys].sort((a, b) => {
|
|
649
|
-
const innerA = a.replace(/^\[|\]
|
|
650
|
-
const innerB = b.replace(/^\[|\]
|
|
843
|
+
const innerA = a.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
|
|
844
|
+
const innerB = b.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
|
|
651
845
|
const matchA = innerA.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
|
|
652
846
|
const matchB = innerB.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
|
|
653
847
|
if (matchA && matchB) {
|
|
@@ -665,7 +859,7 @@ const PROFILE_JARGON = {
|
|
|
665
859
|
const regexParts = [];
|
|
666
860
|
|
|
667
861
|
sortedKeys.forEach(k => {
|
|
668
|
-
const inner = k.replace(/^\[|\]
|
|
862
|
+
const inner = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
|
|
669
863
|
const match = inner.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
|
|
670
864
|
if (match) {
|
|
671
865
|
const label = match[1];
|
|
@@ -673,7 +867,8 @@ const PROFILE_JARGON = {
|
|
|
673
867
|
const aliases = getLabelAliases(label);
|
|
674
868
|
|
|
675
869
|
const escapedAliases = aliases.map(a => a.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
|
676
|
-
const
|
|
870
|
+
const aliasesGroup = `(?:${escapedAliases.join('|')})`;
|
|
871
|
+
const loosePattern = `(?:\\[\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*\\]|<\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*>|\\{\\{\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*\\}\\}|__\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*__|(?<![A-Za-z0-9\\u0400-\\u04FF_])${aliasesGroup}[-_\\s]*0*${baseIndex}(?![A-Za-z0-9\\u0400-\\u04FF_]))(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})?`;
|
|
677
872
|
looseRules.push({ token: k, pattern: loosePattern });
|
|
678
873
|
}
|
|
679
874
|
regexParts.push(k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
|
@@ -681,42 +876,157 @@ const PROFILE_JARGON = {
|
|
|
681
876
|
|
|
682
877
|
let compositeRegex = null;
|
|
683
878
|
if (regexParts.length > 0) {
|
|
684
|
-
compositeRegex = new RegExp(`(?:\\b|\\[)?(?:(?:(?<!\\w)|(?<=\\s))(?:${regexParts.join('|')})(?:(?!\\w)|(?=\\s)))(?:\\b|\\])?`, 'g');
|
|
879
|
+
compositeRegex = new RegExp(`(?:\\b|\\[|<|\\{\\{|__)?(?:(?:(?<!\\w)|(?<=\\s))(?:${regexParts.join('|')})(?:(?!\\w)|(?=\\s)))(?:\\b|\\]|>|\\}\\}|__)?(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})?`, 'g');
|
|
685
880
|
}
|
|
686
881
|
|
|
687
882
|
return { compositeRegex, looseRules };
|
|
688
883
|
}
|
|
689
884
|
|
|
885
|
+
function isJsonPayload(str) {
|
|
886
|
+
if (!str || typeof str !== "string") return false;
|
|
887
|
+
const trimmed = str.trim();
|
|
888
|
+
if (!((trimmed.startsWith("{") && trimmed.endsWith("}")) || (trimmed.startsWith("[") && trimmed.endsWith("]")))) {
|
|
889
|
+
return false;
|
|
890
|
+
}
|
|
891
|
+
try {
|
|
892
|
+
JSON.parse(trimmed);
|
|
893
|
+
return true;
|
|
894
|
+
} catch (_) {
|
|
895
|
+
return false;
|
|
896
|
+
}
|
|
897
|
+
}
|
|
898
|
+
|
|
690
899
|
function cleanAIPromptPrefix(text) {
|
|
691
900
|
if (!text) return "";
|
|
692
|
-
let cleaned = text
|
|
901
|
+
let cleaned = text;
|
|
902
|
+
// If text is a full valid JSON object or array, preserve structure
|
|
903
|
+
if (!isJsonPayload(cleaned)) {
|
|
904
|
+
// 1. Strip raw CSS / style blocks leaked from ChatGPT Canvas, web components or stylesheets (handles multi-line, unclosed and variable definitions)
|
|
905
|
+
cleaned = cleaned.replace(/^\s*(?:[.#][a-zA-Z0-9_-]+|\[[a-zA-Z0-9_#.:\-*>=,'"\s]+\]|:is\([^)]+\)|[a-zA-Z0-9_-]+)?\s*\{[^}]*?(?:\}\s*|\n\n+|$)/gi, "");
|
|
906
|
+
cleaned = cleaned.replace(/^[;{} \t\r\n]+/, "");
|
|
907
|
+
cleaned = cleaned.replace(/(?:^|\n)[a-zA-Z0-9_#.:\-*>[\]=\s,'"]+\{[^}]*(--[a-zA-Z0-9_-]+:|color-mix\(|var\()[^}]*\}/g, "");
|
|
908
|
+
}
|
|
909
|
+
// 2. Strip AI author prefixes and platform artifacts
|
|
910
|
+
cleaned = cleaned.replace(/^\s*(?:Claude responded|Claude|ChatGPT|Gemini|Grok|DeepSeek|Kimi|Copilot|Assistant|User)\s*(?::|\bsaid\b|\bresponded\b|(?=\s))\s*/i, "");
|
|
693
911
|
cleaned = cleaned.replace(/^(?:Here (?:is|are) (?:the )?(?:redacted|scrubbed|sanitized|processed|clean|updated|modified) (?:text|output|version|data).*?[:\n]+|\*\*Scrubbed Text\*\*[:\n]+|### Scrubbed Text[:\n]+)/i, '');
|
|
694
912
|
cleaned = cleaned.replace(/^\s*Edit\s*\n+/i, "");
|
|
695
913
|
cleaned = cleaned.replace(/\s*\bEdit\s+in\s+a\s+page\b\s*$/i, "");
|
|
914
|
+
// 3. Strip stray leading colons, semicolons, or separators left by stripped icons/artifact headers
|
|
915
|
+
cleaned = cleaned.replace(/^[:;|\-\—\–]+(?=\n|$)/, "");
|
|
916
|
+
cleaned = cleaned.replace(/^[:;]+\s*/, "");
|
|
696
917
|
return cleaned.trim();
|
|
697
918
|
}
|
|
698
919
|
|
|
920
|
+
function buildFastTokenLookup(sessionMap) {
|
|
921
|
+
const lookup = new Map();
|
|
922
|
+
const customRegexParts = [];
|
|
923
|
+
const keys = Object.keys(sessionMap || {});
|
|
924
|
+
|
|
925
|
+
for (let i = 0; i < keys.length; i++) {
|
|
926
|
+
const k = keys[i];
|
|
927
|
+
const v = sessionMap[k];
|
|
928
|
+
lookup.set(k, v);
|
|
929
|
+
lookup.set(k.toUpperCase(), v);
|
|
930
|
+
|
|
931
|
+
const inner = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
|
|
932
|
+
const match = inner.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
|
|
933
|
+
if (match) {
|
|
934
|
+
const label = match[1];
|
|
935
|
+
const baseIndex = parseInt(match[2], 10);
|
|
936
|
+
const aliases = getLabelAliases(label);
|
|
937
|
+
for (let a = 0; a < aliases.length; a++) {
|
|
938
|
+
const u = aliases[a].toUpperCase();
|
|
939
|
+
lookup.set(u + '_' + baseIndex, v);
|
|
940
|
+
lookup.set(u + '-' + baseIndex, v);
|
|
941
|
+
lookup.set(u + ' ' + baseIndex, v);
|
|
942
|
+
lookup.set(u + baseIndex, v);
|
|
943
|
+
lookup.set('[' + u + '_' + baseIndex + ']', v);
|
|
944
|
+
lookup.set('<' + u + '_' + baseIndex + '>', v);
|
|
945
|
+
lookup.set('{{' + u + '_' + baseIndex + '}}', v);
|
|
946
|
+
lookup.set('__' + u + '_' + baseIndex + '__', v);
|
|
947
|
+
lookup.set('[' + u + ' ' + baseIndex + ']', v);
|
|
948
|
+
lookup.set('[' + u + '-' + baseIndex + ']', v);
|
|
949
|
+
}
|
|
950
|
+
} else {
|
|
951
|
+
customRegexParts.push(k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
|
952
|
+
}
|
|
953
|
+
}
|
|
954
|
+
|
|
955
|
+
let regexStr = '(?:\\[\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*\\]|<\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*>|\\{\\{\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*\\}\\}|__\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*__|(?<=^|[^a-zA-Z0-9_А-Яа-яЁё])[A-Za-z_А-Яа-яЁё]+[-_\\s]*\\d+)';
|
|
956
|
+
if (customRegexParts.length > 0) {
|
|
957
|
+
regexStr = '(?:' + regexStr + '|' + customRegexParts.join('|') + ')';
|
|
958
|
+
}
|
|
959
|
+
const tokenRegex = new RegExp(regexStr + '(?:\'s|’s|s|[а-яёА-ЯЁ]{1,3})?', 'gi');
|
|
960
|
+
|
|
961
|
+
return { lookup, tokenRegex };
|
|
962
|
+
}
|
|
963
|
+
|
|
964
|
+
function resolveTokenValue(rawMatch, targetTokenKey, sessionMap) {
|
|
965
|
+
if (!sessionMap) return undefined;
|
|
966
|
+
if (sessionMap[targetTokenKey] !== undefined) return sessionMap[targetTokenKey];
|
|
967
|
+
if (sessionMap[rawMatch] !== undefined) return sessionMap[rawMatch];
|
|
968
|
+
const cleanRaw = rawMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
969
|
+
for (const k of Object.keys(sessionMap)) {
|
|
970
|
+
const cleanK = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
971
|
+
if (cleanK.toLowerCase() === cleanRaw.toLowerCase()) {
|
|
972
|
+
return sessionMap[k];
|
|
973
|
+
}
|
|
974
|
+
}
|
|
975
|
+
return undefined;
|
|
976
|
+
}
|
|
977
|
+
|
|
699
978
|
function unscrubText(text, sessionMap) {
|
|
700
979
|
let restoredCount = 0;
|
|
701
980
|
let result = text;
|
|
702
|
-
const tokens = Object.keys(sessionMap);
|
|
981
|
+
const tokens = Object.keys(sessionMap || {});
|
|
703
982
|
if (tokens.length === 0) return { text: result, count: 0 };
|
|
704
983
|
|
|
705
984
|
result = cleanAIPromptPrefix(result);
|
|
706
985
|
|
|
986
|
+
if (tokens.length > 50) {
|
|
987
|
+
const { lookup, tokenRegex } = buildFastTokenLookup(sessionMap);
|
|
988
|
+
result = result.replace(tokenRegex, (match) => {
|
|
989
|
+
if (lookup.has(match)) {
|
|
990
|
+
restoredCount++;
|
|
991
|
+
return lookup.get(match);
|
|
992
|
+
}
|
|
993
|
+
const upper = match.toUpperCase();
|
|
994
|
+
if (lookup.has(upper)) {
|
|
995
|
+
restoredCount++;
|
|
996
|
+
return lookup.get(upper);
|
|
997
|
+
}
|
|
998
|
+
const possMatch = match.match(/^([\s\S]+?)('s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
999
|
+
if (possMatch) {
|
|
1000
|
+
const base = possMatch[1];
|
|
1001
|
+
const suffix = possMatch[2];
|
|
1002
|
+
if (lookup.has(base)) {
|
|
1003
|
+
restoredCount++;
|
|
1004
|
+
return lookup.get(base) + suffix;
|
|
1005
|
+
}
|
|
1006
|
+
if (lookup.has(base.toUpperCase())) {
|
|
1007
|
+
restoredCount++;
|
|
1008
|
+
return lookup.get(base.toUpperCase()) + suffix;
|
|
1009
|
+
}
|
|
1010
|
+
}
|
|
1011
|
+
return match;
|
|
1012
|
+
});
|
|
1013
|
+
return { text: result, count: restoredCount };
|
|
1014
|
+
}
|
|
1015
|
+
|
|
707
1016
|
const { compositeRegex, looseRules } = buildRestorationRegexAndRules(sessionMap);
|
|
708
1017
|
|
|
709
1018
|
if (compositeRegex) {
|
|
710
1019
|
result = result.replace(compositeRegex, (match) => {
|
|
711
|
-
const
|
|
712
|
-
const
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
1020
|
+
const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1021
|
+
const suffix = suffixMatch ? suffixMatch[0] : '';
|
|
1022
|
+
const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
|
|
1023
|
+
const cleanMatch = baseMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
1024
|
+
const token = `[${cleanMatch}]`;
|
|
1025
|
+
|
|
1026
|
+
const val = resolveTokenValue(baseMatch, token, sessionMap) ?? resolveTokenValue(baseMatch, cleanMatch, sessionMap);
|
|
1027
|
+
if (val !== undefined) {
|
|
718
1028
|
restoredCount++;
|
|
719
|
-
return
|
|
1029
|
+
return val + suffix;
|
|
720
1030
|
}
|
|
721
1031
|
return match;
|
|
722
1032
|
});
|
|
@@ -725,8 +1035,16 @@ const PROFILE_JARGON = {
|
|
|
725
1035
|
looseRules.forEach(rule => {
|
|
726
1036
|
const rx = new RegExp(rule.pattern, 'gi');
|
|
727
1037
|
result = result.replace(rx, (match) => {
|
|
728
|
-
|
|
729
|
-
|
|
1038
|
+
const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1039
|
+
const suffix = suffixMatch ? suffixMatch[0] : '';
|
|
1040
|
+
const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
|
|
1041
|
+
|
|
1042
|
+
const val = sessionMap[rule.token] ?? resolveTokenValue(baseMatch, rule.token, sessionMap);
|
|
1043
|
+
if (val !== undefined) {
|
|
1044
|
+
restoredCount++;
|
|
1045
|
+
return val + suffix;
|
|
1046
|
+
}
|
|
1047
|
+
return match;
|
|
730
1048
|
});
|
|
731
1049
|
});
|
|
732
1050
|
|
|
@@ -750,21 +1068,50 @@ const PROFILE_JARGON = {
|
|
|
750
1068
|
const tokens = Object.keys(sessionMap || {});
|
|
751
1069
|
if (tokens.length === 0) return { text: result, count: 0 };
|
|
752
1070
|
|
|
1071
|
+
if (tokens.length > 50) {
|
|
1072
|
+
const { lookup, tokenRegex } = buildFastTokenLookup(sessionMap);
|
|
1073
|
+
result = result.replace(tokenRegex, (match) => {
|
|
1074
|
+
let rawVal = null;
|
|
1075
|
+
let suffix = '';
|
|
1076
|
+
if (lookup.has(match)) {
|
|
1077
|
+
rawVal = lookup.get(match);
|
|
1078
|
+
} else if (lookup.has(match.toUpperCase())) {
|
|
1079
|
+
rawVal = lookup.get(match.toUpperCase());
|
|
1080
|
+
} else {
|
|
1081
|
+
const possMatch = match.match(/^([\s\S]+?)('s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1082
|
+
if (possMatch) {
|
|
1083
|
+
const base = possMatch[1];
|
|
1084
|
+
suffix = possMatch[2];
|
|
1085
|
+
if (lookup.has(base)) rawVal = lookup.get(base);
|
|
1086
|
+
else if (lookup.has(base.toUpperCase())) rawVal = lookup.get(base.toUpperCase());
|
|
1087
|
+
}
|
|
1088
|
+
}
|
|
1089
|
+
if (rawVal !== null) {
|
|
1090
|
+
restoredCount++;
|
|
1091
|
+
const escapedVal = escapeHTML(rawVal);
|
|
1092
|
+
const cleanMatch = match.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
1093
|
+
return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(cleanMatch)}">${escapedVal}</span>${escapeHTML(suffix)}`;
|
|
1094
|
+
}
|
|
1095
|
+
return match;
|
|
1096
|
+
});
|
|
1097
|
+
return { text: result, count: restoredCount };
|
|
1098
|
+
}
|
|
1099
|
+
|
|
753
1100
|
const { compositeRegex, looseRules } = buildRestorationRegexAndRules(sessionMap);
|
|
754
1101
|
|
|
755
1102
|
if (compositeRegex) {
|
|
756
1103
|
result = result.replace(compositeRegex, (match) => {
|
|
757
|
-
const
|
|
758
|
-
const
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
if (
|
|
1104
|
+
const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1105
|
+
const suffix = suffixMatch ? suffixMatch[0] : '';
|
|
1106
|
+
const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
|
|
1107
|
+
const cleanMatch = baseMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
1108
|
+
const token = `[${cleanMatch}]`;
|
|
1109
|
+
|
|
1110
|
+
const val = resolveTokenValue(baseMatch, token, sessionMap) ?? resolveTokenValue(baseMatch, cleanMatch, sessionMap);
|
|
1111
|
+
if (val !== undefined) {
|
|
765
1112
|
restoredCount++;
|
|
766
|
-
const escapedVal = escapeHTML(
|
|
767
|
-
return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(
|
|
1113
|
+
const escapedVal = escapeHTML(val);
|
|
1114
|
+
return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(token)}">${escapedVal}</span>${escapeHTML(suffix)}`;
|
|
768
1115
|
}
|
|
769
1116
|
return match;
|
|
770
1117
|
});
|
|
@@ -778,9 +1125,17 @@ const PROFILE_JARGON = {
|
|
|
778
1125
|
const lastClose = before.lastIndexOf('>');
|
|
779
1126
|
if (lastOpen > lastClose) return match;
|
|
780
1127
|
|
|
781
|
-
|
|
782
|
-
const
|
|
783
|
-
|
|
1128
|
+
const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1129
|
+
const suffix = suffixMatch ? suffixMatch[0] : '';
|
|
1130
|
+
const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
|
|
1131
|
+
|
|
1132
|
+
const val = sessionMap[rule.token] ?? resolveTokenValue(baseMatch, rule.token, sessionMap);
|
|
1133
|
+
if (val !== undefined) {
|
|
1134
|
+
restoredCount++;
|
|
1135
|
+
const escapedVal = escapeHTML(val);
|
|
1136
|
+
return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(rule.token)} (Fuzzy Match)">${escapedVal}</span>${escapeHTML(suffix)}`;
|
|
1137
|
+
}
|
|
1138
|
+
return match;
|
|
784
1139
|
});
|
|
785
1140
|
});
|
|
786
1141
|
|
|
@@ -798,6 +1153,7 @@ const PROFILE_JARGON = {
|
|
|
798
1153
|
LABEL_ALIASES,
|
|
799
1154
|
getLabelAliases,
|
|
800
1155
|
PROFILE_ALIAS_MAP,
|
|
1156
|
+
formatToken,
|
|
801
1157
|
buildRestorationRegexAndRules,
|
|
802
1158
|
unscrubText,
|
|
803
1159
|
unscrubTextAsHTML,
|