@privacyscrubber/mcp-server 1.7.6 → 2.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.well-known/mcp/server-card.json +78 -0
- package/LICENSE +21 -0
- package/README.md +5 -5
- package/index.js +172 -16
- package/package.json +1 -1
- package/ps-license-manager.js +10 -16
- package/ps-pii-engine.cjs +516 -130
- package/ps-pii-engine.js +516 -130
- package/scrubber-core.cjs +438 -150
- package/bundle.mcpb +0 -0
package/ps-pii-engine.cjs
CHANGED
|
@@ -17,39 +17,45 @@ let DEVOPS_SECRETS = [
|
|
|
17
17
|
// Secrets & API Keys
|
|
18
18
|
{ name: 'AWS Credentials', type: 'SECRET', regex: /\b(?:AKIA|ASIA|AGPA|AIDA|AROA|AIPA)[A-Z0-9]{16}\b/g },
|
|
19
19
|
{ name: 'JSON Web Token (JWT)', type: 'SECRET', regex: /\beyJ[a-zA-Z0-9_-]+\.[a-zA-Z0-9_-]+\.[a-zA-Z0-9_-]+\b/g },
|
|
20
|
-
{ name: 'API Token/Key (GitHub/Slack/NPM)', type: 'SECRET', regex: /\b(?:ghp|gho|ghu|ghs|ghr|glpat|npm|xox[baprs])[-_][A-Za-z0-9_]{10,}\b/g },
|
|
21
|
-
{ name: 'Stripe API Key', type: 'SECRET', regex: /\b(?:[rs]k)_(?:test|live)_[a-zA-Z0-9]{
|
|
22
|
-
{ name: '
|
|
20
|
+
{ name: 'API Token/Key (GitHub/Slack/NPM)', type: 'SECRET', regex: /\b(?:ghp|gho|ghu|ghs|ghr|glpat|npm|xox[baprs])[-_][A-Za-z0-9_-]{10,}\b/g },
|
|
21
|
+
{ name: 'Stripe API Key', type: 'SECRET', regex: /\b(?:[rs]k)_(?:test|live)_[a-zA-Z0-9]{14,}\b/g },
|
|
22
|
+
{ name: 'OpenAI Project API Key', type: 'SECRET', regex: /\b(?:sk|pk)-(?:proj-)?[a-zA-Z0-9_-]{16,}\b/gi },
|
|
23
|
+
{ name: 'Database Connection URI', type: 'SECRET', regex: /\b(?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp|mssql):\/\/[^\s"']+/gi },
|
|
24
|
+
{ name: 'Generic Secret/Key', type: 'SECRET', regex: /\b(?:sk|pk|secret|key|token|auth)(?:[-_][a-zA-Z0-9_-]{3,}|(?=[a-zA-Z0-9_-]{5,}\b)(?=[a-zA-Z_-]*[0-9])[a-zA-Z0-9_-]{5,})\b/gi },
|
|
23
25
|
{ name: 'Hash / Hex Key (32-64 chars)', type: 'SECRET', regex: /\b[a-fA-F0-9]{32,64}\b/g },
|
|
24
26
|
{ name: 'CVE Identifier', type: 'SECRET', regex: /\bCVE-\d{4}-\d{4,}\b/gi },
|
|
25
|
-
{ name: 'Cryptographic Hash', type: 'SECRET', regex: /\b(MD5|SHA1|SHA256)[:\s][a-f0-9]{32,64}\b/gi },
|
|
26
|
-
{ name: 'Database/API Secret', type: 'SECRET', regex: /\b(DB|POSTGRES|REDIS|MYSQL|AWS|SECRET|PASSWORD|TOKEN|API|KEY)[A-Z0-9_]*\s*[:=]\s*[^\s"']+\b/gi },
|
|
27
|
-
{ name: 'Proprietary IP / Confidential', type: 'SECRET', regex: /\b(CONFIDENTIAL|PROPRIETARY|TRADE SECRET|DO NOT DISTRIBUTE|INTERNAL USE ONLY)\b/gi },
|
|
27
|
+
{ name: 'Cryptographic Hash', type: 'SECRET', regex: /\b(?:MD5|SHA1|SHA256)[:\s][a-f0-9]{32,64}\b/gi },
|
|
28
|
+
{ name: 'Database/API Secret', type: 'SECRET', regex: /\b(?:DB|POSTGRES|REDIS|MYSQL|AWS|SECRET|PASSWORD|TOKEN|API|KEY)[A-Z0-9_]*\s*[:=]\s*[^\s"']+\b/gi },
|
|
29
|
+
{ name: 'Proprietary IP / Confidential', type: 'SECRET', regex: /\b(?:CONFIDENTIAL|PROPRIETARY|TRADE SECRET|DO NOT DISTRIBUTE|INTERNAL USE ONLY)\b/gi },
|
|
28
30
|
{ name: 'Private Cryptographic Key', type: 'SECRET', regex: /-----BEGIN (?:RSA |EC |PGP |DSA )?PRIVATE KEY-----/g }
|
|
29
31
|
];
|
|
30
32
|
|
|
31
33
|
let REGEX_RULES = [
|
|
32
34
|
...DEVOPS_SECRETS,
|
|
33
35
|
// Emails
|
|
34
|
-
{ type: 'EMAIL', regex:
|
|
36
|
+
{ type: 'EMAIL', regex: /\b[a-zA-Z0-9._%+-]{1,64}@[a-zA-Z0-9.-]{1,255}\.[a-zA-Z]{2,}\b/g },
|
|
35
37
|
|
|
36
|
-
// Financial Data
|
|
37
|
-
{ type: 'FINANCIAL', regex:
|
|
38
|
-
{ type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9]([A-Z0-9]{3})?\b/g },
|
|
39
|
-
{ type: 'FINANCIAL', regex: /\b\d{9}\b/g },
|
|
38
|
+
// Financial Data (PCI-DSS, Bank Accounts, Direct Deposits, Cards, IBAN, SWIFT, Routing Numbers)
|
|
39
|
+
{ type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9](?:[A-Z0-9]{3})?\b/g },
|
|
40
40
|
{ type: 'FINANCIAL', regex: /\b[A-Z]{2}[0-9]{2}[a-zA-Z0-9]{4}[0-9]{7}[a-zA-Z0-9]{0,16}\b/g },
|
|
41
41
|
{ type: 'FINANCIAL', regex: /\bPORTFOLIO[-_][A-Z0-9]{5,}\b/gi },
|
|
42
42
|
{ type: 'FINANCIAL', regex: /\b(?:\d[ -]?){13,19}\b/g },
|
|
43
43
|
{ type: 'FINANCIAL', regex: /\b(?:1|3|bc1)[a-zA-HJ-NP-Z0-9]{25,39}\b/g },
|
|
44
44
|
{ type: 'FINANCIAL', regex: /\b0x[a-fA-F0-9]{40}\b/g },
|
|
45
|
+
// Masked / Direct Deposit Bank Account Numbers & Routing Numbers
|
|
46
|
+
{ type: 'FINANCIAL', regex: /\b(?:Account|Acct|Checking|Savings|Direct\s+Deposit)\s*(?:#|ID|No\.?|Number)?[:\s#]*(?:[\*xX•.-]{3,}\d{2,6}|\d{4}[-\s]?\d{4}[-\s]?\d{2,6})\b/gi },
|
|
47
|
+
{ type: 'FINANCIAL', regex: /\b(?:ABA|Routing|RTN)\s*(?:#|ID|No\.?|Number)?[:\s#]*\d{9}\b/gi },
|
|
45
48
|
|
|
46
49
|
// Legal & Court
|
|
47
|
-
{ type: 'LEGAL', regex: /\bCASE[-_][A-Z0-
|
|
48
|
-
{ type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-
|
|
49
|
-
{ type: 'LEGAL', regex: /\b[A-Z]{2,4}[- ]?\d{2}[- ]?\d{4,}\b/g },
|
|
50
|
+
{ type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9_-]{4,}\b/gi },
|
|
51
|
+
{ type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9_-]{4,}\b/gi },
|
|
50
52
|
{ type: 'PRIVILEGE', regex: /ATTORNEY[- ]CLIENT[- ]PRIVILEGE/gi },
|
|
51
53
|
|
|
52
|
-
// Professional IDs
|
|
54
|
+
// Professional IDs & Organizations
|
|
55
|
+
{ type: 'NAME', isContextName: true, regex: /\b(?:[A-Z][A-Za-z0-9&.,'-]*[ \t\xA0]+){1,5}(?:Inc\.?|LLC|Corp\.?|Corporation|Ltd\.?|Limited|Co\.?|Company|Group|Holdings|Solutions|Services|Technologies|Logistics|Industries|Capital|Bank|Partners|LLP|PLLC)(?:\s+(?:LLC|Inc\.?|Corp\.?|Ltd\.?|USA|Group))?\b/g },
|
|
56
|
+
{ type: 'ID', regex: /\b(?:Employee|Emp|EE|Worker|Staff|File|Badge|Member|Advisor|Producer|Agent|Borrower)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Z0-9-]{3,15}\b/gi },
|
|
57
|
+
{ type: 'ID', regex: /\b(?:Pay\s+Group|Cost\s+Center|Dept|Department)[:\s#]*[A-Za-z0-9_-]{2,30}\b/gi },
|
|
58
|
+
{ type: 'ID', regex: /(?:\b(?:Box\s+d\b|d\.\s*(?:Control|#)?|d\s+Control)\s*(?:number|no\.?|#|num)?[:\s#]*|\bControl\s*(?:number|no\.?|#|num)[:\s#]*|\bControl[:#]\s*)([A-Za-z0-9-]{3,30})/gi },
|
|
53
59
|
{ type: 'ID', regex: /\bEEID[ -]?\d{4,}\b/gi },
|
|
54
60
|
{ type: 'ID', regex: /\bRESUME[-_]?[A-Z0-9]{4,}\b/gi },
|
|
55
61
|
{ type: 'ID', regex: /\bLEAD[-_][A-Z0-9]{5,}\b/gi },
|
|
@@ -69,34 +75,52 @@ let REGEX_RULES = [
|
|
|
69
75
|
{ type: 'ID', regex: /\bCOURSE[-_][A-Z0-9]{4,}\b/gi },
|
|
70
76
|
{ type: 'ID', regex: /\bINSTANCE[-_]ID[-_][a-z0-9-]{10,}\b/gi },
|
|
71
77
|
{ type: 'ID', regex: /\bENV[-_][A-Z0-9]{3,}\b/gi },
|
|
72
|
-
{ type: '
|
|
78
|
+
{ type: 'ID', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
|
|
73
79
|
|
|
74
|
-
//
|
|
75
|
-
{ type: '
|
|
76
|
-
{ type: '
|
|
80
|
+
// Insurance & Health Plan IDs
|
|
81
|
+
{ type: 'ID', regex: /\b(?:BCB|BCBS|AETNA|CIGNA|UHC|HUMANA|MEDICARE|MEDICAID)[-_A-Za-z0-9]+\b/gi },
|
|
82
|
+
{ type: 'ID', regex: /\b(?:Insurance\s+(?:ID|No\.?|Number|#)|Policy(?:\s*(?:ID|No\.?|Number|#)|[:#])|Member\s*(?:ID|No\.?|Number|#|[:#])|Subscriber\s*(?:ID|No\.?|Number|#|[:#])|Group\s*(?:ID|No\.?|Number|#|[:#])|Plan\s*(?:ID|No\.?|Number|#|[:#])|Health(?:\s+Plan)?\s*(?:ID|No\.?|Number|#)|Rx\s*(?:ID|No\.?|Number|Group|BIN|PCN|#))[:\s#]*([A-Za-z0-9-]+)/gi },
|
|
83
|
+
{ type: 'ID', regex: /\b(?:Health\s+Plan(?:\s+Beneficiary)?|Beneficiary(?:\s+(?:No\.?|Number|ID|Num|#))?|HPN)[:\s#]+([A-Za-z0-9-]+)/gi },
|
|
84
|
+
{ type: 'ID', regex: /\bHPN[-_][A-Za-z0-9-]+\b/gi },
|
|
77
85
|
|
|
86
|
+
// Addresses & Locations
|
|
87
|
+
{ type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:\bBox\s+f\b|\bf\.\s*|\bf\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?(?:address[,\s]+and\s+ZIP\s+code|address)?|(?:Employee(?:'s)?\s+address[,\s]+and\s+ZIP\s+code))[\s:#]*([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|Box|\d+\b|1\b|2\b|Wages|Federal|Social|Medicare)/gi },
|
|
88
|
+
{ type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:Borrower(?:'s)?|Co-Borrower(?:'s)?|Employee(?:'s)?|Employer(?:'s)?|Home|Mailing|Property|Physical)\s+address)[\s:#]+([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|City|State|ZIP|SSN|EIN|Phone|Box|\d+\b)/gi },
|
|
89
|
+
{ type: 'ADDRESS', regex: /\b\d{1,6}[ \t\xA0]+(?:[A-Za-z0-9.-]+[ \t\xA0]+){1,4}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Hwy|Highway|Cir|Circle|Trl|Trail|Loop|Row|Pike|Box|PO Box|P\.O\.[ \t\xA0]*Box)\b(?:[ \t\xA0]*,?[ \t\xA0]*(?:Apt|Apartment|Suite|Ste|Unit|#|Fl|Floor|Bldg|Building)\.?[ \t\xA0]*[A-Za-z0-9-]+)?/gi },
|
|
90
|
+
{ type: 'ADDRESS', regex: /\b(?:P\.?O\.?[ \t\xA0]*Box|PO[ \t\xA0]*Box)[ \t\xA0]+\d{1,6}\b/gi },
|
|
91
|
+
{ type: 'ADDRESS', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
|
|
92
|
+
{ type: 'ADDRESS', regex: /\b(?:ZIP|Postal|Code)?\s*(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
|
|
93
|
+
{ type: 'ADDRESS', regex: /\b\d{5}-\d{4}\b/g },
|
|
94
|
+
{ type: 'LOCATION', regex: /\b[A-Za-z][a-zA-Z .'-]{1,25}(?:,\s*(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)|\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR))\b/g },
|
|
78
95
|
|
|
79
96
|
// PHI & Medical
|
|
80
|
-
{ type: 'PHI', regex: /\
|
|
97
|
+
{ type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record (?:No\.?|Num(?:ber)?|#)|Patient (?:No\.?|Num(?:ber)?|#))[\s:#]+([A-Za-z0-9-]+)/gi },
|
|
98
|
+
{ type: 'PHI', regex: /\bMRN[-_ ]*[A-Za-z0-9-]{4,}\b/gi },
|
|
99
|
+
{ type: 'ID', regex: /\b(?:NPI|National Provider Identifier)[:\s#]*(\d{10})\b/gi },
|
|
100
|
+
{ type: 'ID', regex: /\b(?:Device\s+(?:Identifier|ID|Serial|No\.?|Number)|UDI)[:\s#]+([A-Za-z0-9-]+)/gi },
|
|
101
|
+
{ type: 'ID', regex: /\bUDI[-_][A-Za-z0-9-]+\b/gi },
|
|
102
|
+
{ type: 'ID', regex: /\b(?:Vehicle\s+(?:Serial|ID|Identification(?:\s+Number)?|No\.?|Number)|VIN)[:\s#]+([A-Za-z0-9-]+)/gi },
|
|
103
|
+
{ type: 'ID', regex: /\bVIN[-_][A-Za-z0-9-]+\b/gi },
|
|
81
104
|
{ type: 'PHI', regex: /\b[A-TV-Z]\d{2}[. ]?\d[A-Z0-9]?\b/g },
|
|
82
105
|
{ type: 'PHI', regex: /\b[A-Z]{2,3}\d{6,8}\b/g },
|
|
83
106
|
{ type: 'PHI', regex: /\bNHS[ -]?\d{3}[ -]?\d{3}[ -]?\d{4}\b/gi },
|
|
84
107
|
|
|
85
|
-
|
|
86
108
|
// Copyrights
|
|
87
109
|
{ type: 'COPYRIGHT', regex: /\bPROJECT[-_][A-Z0-9]{5,}\b/gi },
|
|
88
110
|
{ type: 'COPYRIGHT', regex: /\b(DRAFT|ASSET|SCRIPT)[-_][0-9]{4,}\b/gi },
|
|
89
111
|
|
|
90
|
-
// General Privacy &
|
|
91
|
-
{ type: '
|
|
92
|
-
{ type: '
|
|
93
|
-
{ type: '
|
|
94
|
-
{ type: '
|
|
95
|
-
{ type: '
|
|
112
|
+
// General Privacy, Dates & Secrets
|
|
113
|
+
{ type: 'ID', regex: /\b(?:GDPR|HIPAA|CCPA|SOC2)[-_]AUDIT[-_]\d{4}\b/gi },
|
|
114
|
+
{ type: 'ID', regex: /\bPOLICY[-_][A-Z0-9]{5,}\b/gi },
|
|
115
|
+
{ type: 'ID', regex: /\bGRADE[S]?\s*:\s*[A-DF][+-]?\b/gi },
|
|
116
|
+
{ type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
|
|
117
|
+
{ type: 'SECRET', regex: /\b(?:PASSWORD|PWD|SECRET)\s*[:=]\s*["']?[\S]{4,}["']?/gi },
|
|
96
118
|
|
|
97
119
|
// Standard IDs (SSN, EIN, Passport, VAT)
|
|
98
120
|
{ type: 'ID', regex: /\b\d{3}-\d{2}-\d{4}\b/g },
|
|
99
|
-
{ type: 'ID', regex:
|
|
121
|
+
{ type: 'ID', regex: /\b(?:XXX|xxx|\*\*\*)[ -]?(?:XX|xx|\*\*)[ -]?\d{4}\b/g },
|
|
122
|
+
{ type: 'ID', regex: /\b\d{2}-\d{7}\b/g },
|
|
123
|
+
{ type: 'ID', regex: /(?:(?:[A-Za-z]:\\|\/(?:usr|var|etc|home|root|Users|private|tmp|opt|bin|sbin|dev|Applications|Library)\/)[a-zA-Z0-9_.-]+(?:[\/\\][a-zA-Z0-9_.-]+)*|\/(?:[a-zA-Z0-9_.-]+\/)+[a-zA-Z0-9_.-]+\.(?:txt|pdf|docx|xlsx|csv|js|ts|json|env|log|key|pem|crt|conf|yaml|yml|xml|html|sql|py|go|rs|c|cpp|h|sh|bin|zip|tar|gz|png|jpg|jpeg|svg|webp|wasm)\b)/g },
|
|
100
124
|
{ type: 'ID', regex: /\b[A-CEGHJ-PR-TW-Z]{1}[A-CEGHJ-NPR-TW-Z]{1}[0-9]{6}[A-DFM]{1}\b/gi },
|
|
101
125
|
{ type: 'ID', regex: /\b[A-Z]{2}[0-9]{6,12}\b/gi },
|
|
102
126
|
{ type: 'ID', regex: /[A-Z0-9<]{30,44}/g },
|
|
@@ -105,14 +129,13 @@ let REGEX_RULES = [
|
|
|
105
129
|
{ type: 'IP', regex: /\b(?:\d{1,3}\.){3}\d{1,3}\b/g },
|
|
106
130
|
{ type: 'IP', regex: /\b(?:[a-fA-F0-9]{1,4}:){7}[a-fA-F0-9]{1,4}\b/g },
|
|
107
131
|
{ type: 'ID', regex: /\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b/g },
|
|
108
|
-
{ type: 'ID', regex: /(?:\B\/|\b[a-zA-Z]:\\)(?:[\w.-]+[\/\\]
|
|
132
|
+
{ type: 'ID', regex: /(?:\B\/|\b[a-zA-Z]:\\)(?:[\w.-]+[\/\\])*[\w.-]+\b/g },
|
|
109
133
|
{ type: 'ID', regex: /\b\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}(?::\d{2})?)?\b/g },
|
|
110
134
|
{ type: 'ID', regex: /\b\d{4}-\d{2}-\d{2}\b/g },
|
|
111
135
|
{ type: 'ID', regex: /\b\d{2}\/\d{2}\/\d{4}\b/g },
|
|
112
136
|
|
|
113
|
-
// Phone Numbers
|
|
114
|
-
{ type: 'PHONE', regex:
|
|
115
|
-
{ type: 'PHONE', regex: /(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}/g },
|
|
137
|
+
// Phone Numbers
|
|
138
|
+
{ type: 'PHONE', regex: /(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}\b/g },
|
|
116
139
|
{ type: 'PHONE', regex: /\+?[1-9]\d{1,3}[\s.-]\(?\d{1,4}\)?[\s.-]\d{2,4}[\s.-]\d{4}/g },
|
|
117
140
|
{ type: 'PHONE', regex: /(?:\+44\s?7\d{3}|\(?07\d{3}\)?)\s?\d{3}\s?\d{3}\b/g },
|
|
118
141
|
{ type: 'PHONE', regex: /\b(?:\d{3}[-.\s]\d{4}|\(\d{3}\)\s??\d{3}[-.\s]??\d{4}|\d{3}[-.\s]??\d{3}[-.\s]??\d{4})\b/g },
|
|
@@ -126,50 +149,70 @@ let REGEX_RULES = [
|
|
|
126
149
|
{ type: 'ID', regex: /\bDRIVER[S]?\s+LICENSE[ -]?\d{6,15}\b/gi },
|
|
127
150
|
|
|
128
151
|
// Generalized Name Detection (First [Middle] Last) — max 1 middle word to avoid grabbing job titles
|
|
129
|
-
// Middle word must be: a particle (van/de/etc.), a single initial (A.), or a capitalized word of ≥2 lowercase letters
|
|
130
|
-
{ type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}[\p{Ll}'-]*
|
|
131
|
-
//
|
|
132
|
-
{ type: 'NAME', regex:
|
|
133
|
-
//
|
|
134
|
-
|
|
135
|
-
|
|
152
|
+
// Middle word must be: a particle (van/de/etc.), a single initial (A. or A), or a capitalized word of ≥2 lowercase letters
|
|
153
|
+
{ type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*|\p{Lu}\.?|van|von|de|di|da|la|le|del|du|der|van[ \t\xA0]+de|van[ \t\xA0]+der)){0,1}[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
154
|
+
// Payroll Format Names (e.g. "BARKER, KELLY", "BARKER, KELLY M", "DOE, JOHN M.")
|
|
155
|
+
{ type: 'NAME', regex: /\b[A-Z]{2,25},\s+[A-Z]{2,25}(?:\s+[A-Z]\.?|\s+[A-Z]{2,25})*\b/g },
|
|
156
|
+
// All-Caps Names (2–3 words, supporting single-letter middle initial e.g. "KELLY M BARKER", "JOHN M BARKER")
|
|
157
|
+
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+(?:\p{Lu}\.?|[A-Z][a-z]+))?[ \t\xA0]+\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
158
|
+
// ALL-CAPS first + middle initial(s) with optional spaces + ALL-CAPS last name
|
|
159
|
+
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
136
160
|
// ALL-CAPS first name + optional middle initial(s) with optional spaces + Mixed-Case last name
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])(?:Mr|Mrs|Ms|Dr|Prof|Hon|Mr\.|Mrs\.|Ms\.|Dr\.|Prof\.|Hon\.)[ \t\xA0]+\p{Lu}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
161
|
+
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*|[ \t\xA0]+)(?:\p{Lu}\p{Ll}[\p{Ll}'-]*|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
162
|
+
// Names with Honorifics (with or without period, supporting single or multi-word full names) (Unicode-safe)
|
|
163
|
+
{ type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])(?:Mr|Mrs|Ms|Dr|Prof|Hon|Mr\.|Mrs\.|Ms\.|Dr\.|Prof\.|Hon\.)[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*))?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
141
164
|
|
|
142
|
-
//
|
|
143
|
-
{ type: 'NAME', isContextName: true, regex: /(?:
|
|
165
|
+
// W-2 Box c Employer Block (Name, Address, and Zip Code)
|
|
166
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:\bBox\s+c\b|\bc\.\s*|\bc\s+(?=Employer))\s*(?:Employer(?:'s)?\s*)?(?:name[,\s]+address[,\s]+and\s+ZIP\s+code|name)?|(?:Employer(?:'s)?\s+name[,\s]+address[,\s]+and\s+ZIP\s+code))[\s:#]*([A-Za-z0-9&., \t\xA0'-]{2,45}?)(?=\r?\n|$|\s{3,}|\t|EIN|FEIN|Box|\d+\b|Wages|Federal|Social|Medicare)/gi },
|
|
167
|
+
// W-2 Box e Employee Name & Initial
|
|
168
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:\bBox\s+e\b|\be\.\s*|\be\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?(?:first\s+name(?:\s+(?:and|&)\s+initial)?|name)?[:\s#]*|(?:Employee(?:'s)?\s+first\s+name(?:\s+(?:and|&)\s+initial)?))[\s:#]*([A-Za-z0-9.\s'-]{2,35}?)(?=\r?\n|$|\s{3,}|\t|Last|Surname|Suff|Box|\d+\b|1\b)/gi },
|
|
169
|
+
// Contextual First Names (Employee's first name, First name, Given name)
|
|
170
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:Employee(?:'s)?|Borrower(?:'s)?|Co-Borrower(?:'s)?|Applicant(?:'s)?|Candidate(?:'s)?|Worker(?:'s)?|Taxpayer(?:'s)?|Spouse(?:'s)?|Person(?:'s)?)\s+)?(?:First\s+name(?:\s+(?:and|&)\s+initial)?|Given\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9.\s'-]{2,30}?)(?=\r?\n|$|\s{3,}|\t|Last|Surname|Family|Suff|Box|Address|SSN|EIN)/gi },
|
|
171
|
+
// Contextual Last Names (Last name, Surname, Family name)
|
|
172
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:Employee(?:'s)?|Borrower(?:'s)?|Co-Borrower(?:'s)?|Applicant(?:'s)?|Candidate(?:'s)?|Worker(?:'s)?|Taxpayer(?:'s)?|Spouse(?:'s)?|Person(?:'s)?)\s+)?(?:Last\s+name|Surname|Family\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z'-]{2,30})/gi },
|
|
173
|
+
// Contextual General Names (Employee, Borrower, Co-Borrower, Taxpayer, Spouse, Applicant, Candidate, Worker, Employer, Company, Insured, Patient, Client, etc.)
|
|
174
|
+
{ type: 'NAME', isContextName: true, regex: /(?:Employee(?:\s+Name)?|Employer(?:\s+Name)?|Borrower(?:\s+Name)?|Co-Borrower(?:\s+Name)?|Applicant(?:\s+Name)?|Candidate(?:\s+Name)?|Worker(?:\s+Name)?|Taxpayer(?:\s+Name)?|Spouse(?:\s+Name)?|Manager|Supervisor|Reporting To|Insured|Claimant|Patient|Client|Customer|Account Holder|Prepared By|Attention|Attn|Contact(?: Name)?|Child|Parent|Guardian|Relationship|Kin|Tenant|Landlord|Buyer|Seller|Plaintiff|Defendant|Testator)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9&.,\s'-]{2,40}?)(?=\r?\n|$|\s{3,}|\t|Employee|Employer|Address|Phone|SSN|EIN|FEIN|Date|Pay|Rate|Tax|W-2|OMB|Copy|Box|Status)/gi },
|
|
175
|
+
// Box e shorthand
|
|
176
|
+
{ type: 'NAME', isContextName: true, regex: /\b(?:Box\s+e)\s*[:#-]\s*([A-Za-z0-9&.,\s'-]{2,40})/gi }
|
|
144
177
|
];
|
|
145
178
|
|
|
146
179
|
let PROFILE_RULES = {
|
|
147
180
|
general: [],
|
|
148
181
|
legal: [
|
|
149
|
-
{ type: 'LEGAL', regex: /\bCASE[-_][A-Z0-
|
|
150
|
-
{ type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-
|
|
182
|
+
{ type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9_-]{4,}\b/gi },
|
|
183
|
+
{ type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9_-]{4,}\b/gi },
|
|
151
184
|
{ type: 'LEGAL', regex: /\b[A-Z]{2,4}[- ]?\d{2}[- ]?\d{4,}\b/g },
|
|
152
185
|
{ type: 'PRIVILEGE', regex: /ATTORNEY[- ]CLIENT[- ]PRIVILEGE/gi }
|
|
153
186
|
],
|
|
154
187
|
hr: [
|
|
155
|
-
{ type: 'NAME', isContextName: true, regex: /(?:Candidate|Applicant|Employee|Reporting To|Manager)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g },
|
|
156
|
-
|
|
188
|
+
{ type: 'NAME', isContextName: true, regex: /(?:Candidate|Applicant|Employee|Reporting To|Manager|Mentored by|Direct Report)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g },
|
|
157
189
|
{ type: 'ID', regex: /\bEEID[ -]?\d{4,}\b/gi },
|
|
158
190
|
{ type: 'ID', regex: /\bEMP[-_]\d{3,}\b/gi },
|
|
159
191
|
{ type: 'ID', regex: /\bRESUME[-_]?[A-Z0-9]{4,}\b/gi },
|
|
160
|
-
{ type: '
|
|
161
|
-
{ type: '
|
|
192
|
+
{ type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
|
|
193
|
+
{ type: 'DATE', regex: /\b(?:Graduated|Graduation|Class of)[:\s]+(?:(?:Spring|Summer|Fall|Winter)\s+)?(?:(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec|January|February|March|April|June|July|August|September|October|November|December)\s+)?\d{4}\b/gi },
|
|
194
|
+
{ type: 'ID', regex: /(?:https?:\/\/)?(?:www\.)?linkedin\.com\/in\/[A-Za-z0-9_-]+/gi },
|
|
195
|
+
{ type: 'ID', regex: /(?:https?:\/\/)?(?:www\.)?github\.com\/[A-Za-z0-9_-]+/gi },
|
|
196
|
+
{ type: 'ADDRESS', regex: /\b\d{1,6}\s+(?:[A-Z0-9][a-zA-Z0-9-]*\s+){1,3}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Highway|Hwy|Circle|Cir|Trail|Trl|Dr(?!\.?\s+[A-Z][a-z]+))\b/g }
|
|
162
197
|
],
|
|
163
198
|
finance: [
|
|
164
|
-
{ type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9]([A-Z0-9]{3})?\b/g },
|
|
199
|
+
{ type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9](?:[A-Z0-9]{3})?\b/g },
|
|
165
200
|
{ type: 'FINANCIAL', regex: /\b[A-Z]{2}[0-9]{2}[a-zA-Z0-9]{4}[0-9]{7}[a-zA-Z0-9]{0,16}\b/g },
|
|
166
201
|
{ type: 'FINANCIAL', regex: /\bPORTFOLIO[-_][A-Z0-9]{5,}\b/gi }
|
|
167
202
|
],
|
|
168
203
|
medical: [
|
|
169
|
-
{ type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record No|Patient No)[\s:#]
|
|
170
|
-
{ type: '
|
|
171
|
-
|
|
172
|
-
{ type: '
|
|
204
|
+
{ type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record (?:No\.?|Num(?:ber)?|#)|Patient (?:No\.?|Num(?:ber)?|#))[\s:#]+([A-Za-z0-9-]+)/gi },
|
|
205
|
+
{ type: 'DATE', regex: /\b(?:DOB|Date of Birth|BIRTHDAY)[\s:]+([0-9./-]{6,10})\b/gi },
|
|
206
|
+
{ type: 'PHI', regex: /\bMRN[-_ ]*[A-Za-z0-9-]{4,}\b/gi },
|
|
207
|
+
{ type: 'ID', regex: /\b(?:Insurance\s+(?:ID|No\.?|Number|#)|Policy(?:\s*(?:ID|No\.?|Number|#)|[:#])|Member\s*(?:ID|No\.?|Number|#|[:#])|Subscriber\s*(?:ID|No\.?|Number|#|[:#])|Group\s*(?:ID|No\.?|Number|#|[:#])|Plan\s*(?:ID|No\.?|Number|#|[:#])|Health(?:\s+Plan)?\s*(?:ID|No\.?|Number|#)|Rx\s*(?:ID|No\.?|Number|Group|BIN|PCN|#))[:\s#]*([A-Za-z0-9-]+)/gi },
|
|
208
|
+
{ type: 'ID', regex: /\b(?:Health\s+Plan(?:\s+Beneficiary)?|Beneficiary(?:\s+No\.?|\s+Number)?|HPN)[:\s#]+([A-Za-z0-9-]+)/gi },
|
|
209
|
+
{ type: 'ID', regex: /\bHPN[-_][A-Za-z0-9-]+\b/gi },
|
|
210
|
+
{ type: 'ID', regex: /\b(?:BCB|BCBS|AETNA|CIGNA|UHC|HUMANA|MEDICARE|MEDICAID)[-_A-Za-z0-9]+\b/gi },
|
|
211
|
+
{ type: 'ID', regex: /\b(?:NPI|National Provider Identifier)[:\s#]*(\d{10})\b/gi },
|
|
212
|
+
{ type: 'ID', regex: /\b(?:Device\s+(?:Identifier|ID|Serial|No\.?|Number)|UDI)[:\s#]+([A-Za-z0-9-]+)/gi },
|
|
213
|
+
{ type: 'ID', regex: /\bUDI[-_][A-Za-z0-9-]+\b/gi },
|
|
214
|
+
{ type: 'ID', regex: /\b(?:Vehicle\s+(?:Serial|ID|Identification(?:\s+Number)?|No\.?|Number)|VIN)[:\s#]+([A-Za-z0-9-]+)/gi },
|
|
215
|
+
{ type: 'ID', regex: /\bVIN[-_][A-Za-z0-9-]+\b/gi },
|
|
173
216
|
{ type: 'PHI', regex: /\b[A-TV-Z]\d{2}[. ]?\d[A-Z0-9]?\b/g },
|
|
174
217
|
{ type: 'PHI', regex: /\b[A-Z]{2,3}\d{6,8}\b/g },
|
|
175
218
|
{ type: 'PHI', regex: /\bNHS[ -]?\d{3}[ -]?\d{3}[ -]?\d{4}\b/gi }
|
|
@@ -186,13 +229,13 @@ let PROFILE_RULES = {
|
|
|
186
229
|
bizops: [
|
|
187
230
|
{ type: 'ID', regex: /\b(?:DEAL|KPI|METRIC)[-_: ]?[A-Z0-9]{4,}\b/gi },
|
|
188
231
|
{ type: 'ID', regex: /\b(?:ENTITY|VENDOR|PARTNER)[-_: ]?[0-9]{4,10}\b/gi },
|
|
189
|
-
{ type: 'FINANCIAL', regex: /\b(?:REVENUE|EBITDA|PROFIT|MARGIN)[-
|
|
232
|
+
{ type: 'FINANCIAL', regex: /\b(?:REVENUE|EBITDA|PROFIT|MARGIN)[\s:_-]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
|
|
190
233
|
{ type: 'SECRET', regex: /\b(?:NDA|M&A|MERGER)[-_: ]?[A-Z0-9]{4,}\b/gi }
|
|
191
234
|
],
|
|
192
235
|
sales: [
|
|
193
236
|
{ type: 'ID', regex: /\bOPPORTUNITY[-_: ]?[A-Z0-9]{5,}\b/gi },
|
|
194
237
|
{ type: 'ID', regex: /\b(?:DOCUSIGN|CONTRACT)[-_: ]?[0-9A-F]{8,32}\b/gi },
|
|
195
|
-
{ type: 'FINANCIAL', regex: /\b(?:ARR|MRR|QUOTA)[\s:]
|
|
238
|
+
{ type: 'FINANCIAL', regex: /\b(?:ARR|MRR|QUOTA)[\s:]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
|
|
196
239
|
{ type: 'ID', regex: /\b(?:SFDC|HUBSPOT)[-_: ]?[0-9A-Z]{15,18}\b/gi }
|
|
197
240
|
],
|
|
198
241
|
support: [
|
|
@@ -205,8 +248,8 @@ let PROFILE_RULES = {
|
|
|
205
248
|
{ type: 'ID', regex: /\b(?:MLS|LIS)[- ]?\d{6,10}\b/gi },
|
|
206
249
|
{ type: 'ID', regex: /\bPARCEL[- ]?\d{5,15}\b/gi },
|
|
207
250
|
{ type: 'ID', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
|
|
208
|
-
{ type: 'FINANCIAL', regex: /\b(?:RENT|LEASE|ESCROW)[\s:]
|
|
209
|
-
{ type: 'SECRET', regex: /\b(?:GATE|DOOR|LOBBY)[-_ ](?:CODE|PIN)[\s:]
|
|
251
|
+
{ type: 'FINANCIAL', regex: /\b(?:RENT|LEASE|ESCROW)[\s:]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)[0-9,]{3,}\b/gi },
|
|
252
|
+
{ type: 'SECRET', regex: /\b(?:GATE|DOOR|LOBBY)[-_ ](?:CODE|PIN)[\s:]*\d{4,6}\b/gi }
|
|
210
253
|
],
|
|
211
254
|
compliance: [
|
|
212
255
|
{ type: 'SECRET', regex: /\b(?:GDPR|HIPAA|CCPA|SOC2|ISO27001)[-_: ]?AUDIT[-_: ]?\d{4}\b/gi },
|
|
@@ -214,10 +257,10 @@ let PROFILE_RULES = {
|
|
|
214
257
|
{ type: 'ID', regex: /\b(?:SAR|DSAR)[-_\/: ]?[A-Z0-9-/]+\b/gi }
|
|
215
258
|
],
|
|
216
259
|
ccpa: [
|
|
217
|
-
{ type: 'ID', regex: /\
|
|
260
|
+
{ type: 'ID', regex: /\b(?:DL|DRIVER['’]?S?\s+LICENSE)[:\s#-]*[A-Z0-9]{6,12}\b/gi },
|
|
218
261
|
{ type: 'LOCATION', regex: /\b-?\d{1,3}\.\d{4,6}[° ]?[NSns],\s*-?\d{1,3}\.\d{4,6}[° ]?[EWew]\b/g },
|
|
219
|
-
{ type: '
|
|
220
|
-
{ type: 'ID', regex: /\bACCOUNT[ -]?(?:ID|NUM|NUMBER)[:\s][A-Z0-9]{6,20}\b/gi }
|
|
262
|
+
{ type: 'ID', regex: /\b(?:CCPA|CPRA)[-_: ]?OPT[-_ ]OUT\b/gi },
|
|
263
|
+
{ type: 'ID', regex: /\bACCOUNT[ -]?(?:ID|NUM|NUMBER)[:\s]+[A-Z0-9]{6,20}\b/gi }
|
|
221
264
|
],
|
|
222
265
|
engineering: [
|
|
223
266
|
{ type: 'SECRET', regex: /(?<=\b(?:DB|POSTGRES|REDIS|MYSQL|AWS|SECRET|PASSWORD|TOKEN|API|KEY)[A-Z0-9_]*\s*[:=]\s*["']?)[A-Za-z0-9_-]{10,}/gi },
|
|
@@ -226,7 +269,7 @@ let PROFILE_RULES = {
|
|
|
226
269
|
{ type: 'ID', regex: /\b[a-z0-9](?:[-a-z0-9]*[a-z0-9])?\.svc\.cluster\.local\b/g }
|
|
227
270
|
],
|
|
228
271
|
agents: [
|
|
229
|
-
{ type: 'ID', regex: /\b(?:AGENT|VECTOR|EMBEDDING)[-_: ]?[A-Z0-9]{8,}\b/gi },
|
|
272
|
+
{ type: 'ID', regex: /\b(?:AGENT|VECTOR|EMBEDDING)[-_: ]?(?:ID[-_: ]?)?[A-Z0-9]{8,}\b/gi },
|
|
230
273
|
{ type: 'ID', regex: /\bTASK[-_: ]?[A-Z0-9]{5,15}\b/gi },
|
|
231
274
|
{ type: 'SECRET', regex: /\b(?:SYS_PROMPT|SYSTEM_PROMPT|OPENAI_API_KEY)[-_: ]?[A-Za-z0-9_-]{10,}\b/gi }
|
|
232
275
|
],
|
|
@@ -247,7 +290,7 @@ let PROFILE_RULES = {
|
|
|
247
290
|
{ type: 'SECRET', regex: /\b(?:CONFIG|KUBECONFIG|TFSTATE)[-_: ]?[A-Z0-9]{6,15}\b/gi }
|
|
248
291
|
],
|
|
249
292
|
personal: [
|
|
250
|
-
{ type: '
|
|
293
|
+
{ type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
|
|
251
294
|
{ type: 'SECRET', regex: /\b(?:PASSWORD|PWD|SECRET|PIN)[\s:]*[\S]{4,20}\b/gi },
|
|
252
295
|
{ type: 'PHONE', regex: /\b(?:WIFE|HUSBAND|PARTNER|MOM|DAD)[\s:]+(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}\b/gi }
|
|
253
296
|
],
|
|
@@ -294,10 +337,46 @@ let PROFILE_RULES = {
|
|
|
294
337
|
{ type: 'ID', regex: /\b(?:Batch|Lot|Serial)\s*(?:No[.]?|Number|#)?[:\s#]*[A-Z0-9][A-Z0-9\-]{3,14}\b/gi },
|
|
295
338
|
{ type: 'PHI', regex: /\b(?:Dose|Dosage)[:\s]+\d+(?:\.\d+)?\s*(?:mg|mcg|mL|IU|units?)\b/gi },
|
|
296
339
|
{ type: 'ID', regex: /\b(?:CRF|eCRF|Case\s+Report\s+Form)\s*(?:No|Page|ID)?[:\s#]*[A-Z0-9]{2,10}\b/gi }
|
|
340
|
+
],
|
|
341
|
+
underwriting: [
|
|
342
|
+
// Employer Corporate / Business Names
|
|
343
|
+
{ type: 'NAME', isContextName: true, regex: /\b(?:[A-Z][A-Za-z0-9&.,'-]*[ \t\xA0]+){1,5}(?:Inc\.?|LLC|Corp\.?|Corporation|Ltd\.?|Limited|Co\.?|Company|Group|Holdings|Solutions|Services|Technologies|Logistics|Industries|Capital|Bank|Partners|LLP|PLLC)(?:\s+(?:LLC|Inc\.?|Corp\.?|Ltd\.?|USA|Group))?\b/g },
|
|
344
|
+
{ type: 'NAME', isContextName: true, regex: /(?:Employer|Company|Organization|Business)\s*(?:Name)?[\s:#]+([A-Za-z0-9&.,\s'-]{2,40}?)(?=\r?\n|$|\s{3,}|\t|Address|EIN|FEIN|Phone|W-2|Rate|Pay|Wage)/gi },
|
|
345
|
+
|
|
346
|
+
// W-2 & Tax Identifiers
|
|
347
|
+
{ type: 'ID', regex: /(?:\b(?:Box\s+d\b|d\.\s*(?:Control|#)?|d\s+Control)\s*(?:number|no\.?|#|num)?[:\s#]*|\bControl\s*(?:number|no\.?|#|num)[:\s#]*|\bControl[:#]\s*)([A-Za-z0-9-]{3,30})/gi },
|
|
348
|
+
{ type: 'ID', regex: /\b(?:EIN|FEIN|Tax\s+ID)[:\s#]*\d{2}-\d{7}\b/gi },
|
|
349
|
+
{ type: 'ID', regex: /\b\d{2}-\d{7}\b/g },
|
|
350
|
+
{ type: 'ID', regex: /\b\d{3}-\d{2}-\d{4}\b/g },
|
|
351
|
+
{ type: 'ID', regex: /\b(?:XXX|xxx|\*\*\*)[ -]?(?:XX|xx|\*\*)[ -]?\d{4}\b/g },
|
|
352
|
+
|
|
353
|
+
// Employee, Loan & Payroll IDs
|
|
354
|
+
{ type: 'ID', regex: /\b(?:Employee|Emp|EE|Worker|Borrower|Badge|Advisor|Producer|Agent|Applicant|File)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Z0-9-]{3,20}\b/gi },
|
|
355
|
+
{ type: 'ID', regex: /\b(?:Pay\s+Group|Cost\s+Center|Dept|Department)[:\s#]*[A-Za-z0-9_-]{2,30}\b/gi },
|
|
356
|
+
{ type: 'ID', regex: /\b(?:Loan|Application|Deal|Borrower|File)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Za-z0-9-]{4,25}\b/gi },
|
|
357
|
+
|
|
358
|
+
// Direct Deposit, Bank Accounts & Routing Numbers (Masked & Unmasked)
|
|
359
|
+
{ type: 'FINANCIAL', regex: /\b(?:Account|Acct|Checking|Savings|Direct\s+Deposit)\s*(?:#|ID|No\.?|Number)?[:\s#]*(?:[\*xX•.-]{3,}\d{2,6}|\d{4}[-\s]?\d{4}[-\s]?\d{2,6})\b/gi },
|
|
360
|
+
{ type: 'FINANCIAL', regex: /\b(?:ABA|Routing|RTN)\s*(?:#|ID|No\.?|Number)?[:\s#]*\d{9}\b/gi },
|
|
361
|
+
|
|
362
|
+
// Borrower & Co-Borrower Names (ALL-CAPS, Payroll, Title Case)
|
|
363
|
+
{ type: 'NAME', regex: /\b[A-Z]{2,25},\s+[A-Z]{2,25}(?:\s+[A-Z]\.?|\s+[A-Z]{2,25})*\b/g },
|
|
364
|
+
{ type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+(?:\p{Lu}\.?|[A-Z][a-z]+))?[ \t\xA0]+\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
|
|
365
|
+
{ type: 'NAME', isContextName: true, regex: /(?:(?:Borrower|Co-Borrower|Applicant|Co-Applicant|Employee|Worker|Taxpayer|Candidate|Primary\s+Borrower|Joint\s+Borrower|Account\s+Holder|Insured|Client)\s*(?:Name)?|First\s+name|Given\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9&.,\s'-]{2,35}?)(?=\r?\n|$|\s{3,}|\t|SSN|EIN|DOB|Address|Phone|Rate|Pay|Wage|Date|Box|Last|Surname)/gi },
|
|
366
|
+
{ type: 'NAME', isContextName: true, regex: /(?:Last\s+name|Surname|Family\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z'-]{2,30})/gi },
|
|
367
|
+
|
|
368
|
+
// Addresses & Locations
|
|
369
|
+
{ type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:Borrower(?:'s)?|Co-Borrower(?:'s)?|Employee(?:'s)?|Employer(?:'s)?|Home|Mailing|Property|Physical)\s+address|(?:(?:\bBox\s+f\b|\bf\.\s*|\bf\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?address))[\s:#]+([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|City|State|ZIP|SSN|EIN|Phone|Box|\d+\b)/gi },
|
|
370
|
+
{ type: 'ADDRESS', regex: /\b\d{1,6}[ \t\xA0]+(?:[A-Za-z0-9.-]+[ \t\xA0]+){1,4}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Hwy|Highway|Cir|Circle|Trl|Trail|Loop|Row|Pike|Box|PO Box|P\.O\.[ \t\xA0]*Box)\b(?:[ \t\xA0]*,?[ \t\xA0]*(?:Apt|Apartment|Suite|Ste|Unit|#|Fl|Floor|Bldg|Building)\.?[ \t\xA0]*[A-Za-z0-9-]+)?/gi },
|
|
371
|
+
{ type: 'ADDRESS', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
|
|
372
|
+
{ type: 'ADDRESS', regex: /\b\d{5}-\d{4}\b/g },
|
|
373
|
+
{ type: 'LOCATION', regex: /\b[A-Za-z][a-zA-Z .'-]{1,25}(?:,\s*(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)|\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR))\b/g }
|
|
297
374
|
]
|
|
298
375
|
};
|
|
299
376
|
|
|
300
377
|
let NAME_STOP_LIST = new Set([
|
|
378
|
+
'clinical note', 'case note', 'prod log', 'siem alert', 'critical security incident', 'security incident', 'hr review', 'crm export', 'bank statement', 'file export', 'database row', 'lease application', 'strategy export', 'action log', 'glossary query', 'tool comparison', 'call transcript', 'zendesk ticket', 'board minutes', 'agent context', 'config dump', 'database dump', 'patient note', 'medical record', 'admission note', 'discharge summary', 'progress note', 'hiring review', 'security audit', 'incident response', 'server log', 'system log', 'api response', 'error log', 'audit log', 'debug log',
|
|
379
|
+
'tax statement', 'wage and tax statement', 'wage and tax', 'wage statement', 'earning statement', 'earnings statement', 'pay statement', 'pay stub', 'paystub', 'withholding statement',
|
|
301
380
|
'case no', 'account no', 'client no', 'ref no', 'matter no',
|
|
302
381
|
'affected user', 'incident date', 'incident type', 'incident report',
|
|
303
382
|
'review period', 'review date', 'salary band', 'salary range',
|
|
@@ -333,13 +412,15 @@ let NAME_STOP_LIST = new Set([
|
|
|
333
412
|
'first name', 'last name', 'middle name', 'full name', 'email address', 'phone number', 'cell phone', 'home phone', 'zip code', 'postal code', 'page number', 'section one', 'table contents', 'table of', 'figure one',
|
|
334
413
|
'marketing department', 'sales department', 'engineering team', 'product team', 'customer support', 'human resources', 'public relations',
|
|
335
414
|
'artificial intelligence', 'machine learning', 'deep learning', 'large language', 'operating system', 'source code', 'user interface', 'web browser', 'pull request', 'merge request', 'commit message', 'code review', 'cloud computing', 'database schema',
|
|
336
|
-
'blood pressure', 'heart rate', 'chief physician', 'treating physician', 'health care', 'healthcare provider', 'medical record',
|
|
415
|
+
'blood pressure', 'heart rate', 'chief physician', 'treating physician', 'health care', 'healthcare provider', 'medical record', 'medical record number', 'acute bronchitis', 'vital signs', 'vital sign', 'health plan', 'health plan beneficiary', 'device identifier', 'vehicle serial',
|
|
337
416
|
'grade a', 'grade b', 'grade c', 'grade d', 'grade f',
|
|
338
417
|
'version 1', 'version 2', 'version 3', 'version 4', 'version 5',
|
|
339
418
|
'step 1', 'step 2', 'step 3', 'step 4', 'step 5',
|
|
340
419
|
'page 1', 'page 2', 'page 3', 'page 4', 'page 5',
|
|
341
420
|
'cs101', 'course cs101',
|
|
342
|
-
// Expanded Stop List (Common nouns
|
|
421
|
+
// Expanded Stop List (Common nouns, command phrases, legal, prompt, chess, and animation terms)
|
|
422
|
+
'docket number', 'docket numbers', 'dockets section', 'case name', 'case names', 'case number', 'case numbers', 'law firm', 'law firms', 'counsel stack', 'counselstack', 'counselstack connector', 'tier 0', 'tier 1', 'tier 2', 'tier 3', 'tier 4', 'do not', 'do not write', 'specific permission', 'write again', 'without permission', 'without specific permission', 'on screen', 'in report', 'own line', 'connector access', 'prompt instruction', 'prompt instructions', 'finding report', 'findings report',
|
|
423
|
+
'white bishop', 'black bishop', 'white knight', 'black knight', 'white king', 'black king', 'white queen', 'black queen', 'white rook', 'black rook', 'white pawn', 'black pawn', 'chess piece', 'chess pieces', 'chess game', 'chess match', 'disney-pixar', 'disney pixar', 'pixar animation', 'close-up', 'close up',
|
|
343
424
|
'january', 'february', 'march', 'april', 'may', 'june', 'july', 'august', 'september', 'october', 'november', 'december',
|
|
344
425
|
'monday', 'tuesday', 'wednesday', 'thursday', 'friday', 'saturday', 'sunday',
|
|
345
426
|
'yesterday', 'tomorrow', 'today', 'last week', 'next month', 'early morning', 'late night',
|
|
@@ -373,24 +454,115 @@ let NAME_STOP_LIST = new Set([
|
|
|
373
454
|
'class name', 'function name', 'variable name', 'database table', 'schema name', 'index name', 'query result', 'error message', 'warning message', 'log entry', 'debug log', 'stack trace',
|
|
374
455
|
'staff member', 'team member', 'board member', 'board meeting', 'committee member', 'executive board',
|
|
375
456
|
'email us', 'contact us', 'about us', 'sign in', 'sign out',
|
|
457
|
+
'driver license', 'drivers license', 'opt out', 'opt-out', 'ccpa opt', 'cpra opt', 'spoiler unreleased', 'unreleased draft',
|
|
376
458
|
'lighting', 'keyboard', 'creating', 'building', 'training', 'planning', 'starting', 'painting', 'printing', 'returned', 'released', 'required', 'accepted', 'imported', 'services', 'products', 'accounts', 'settings', 'partners', 'keywords', 'keystone', 'keyspace', 'keynotes', 'keychain'
|
|
377
|
-
, 'quarterly results', 'strategic planning', 'market research', 'customer base', 'privacy settings', 'account settings', 'security settings', 'download now', 'free trial', 'limited time', 'copyright protected', 'all rights', 'rights reserved', 'credit score', 'monthly rent', 'lease application', 'property address', 'reference number', 'additional identifier', 'lease agreement', 'hiring review', 'candidate name', 'privacy policy', 'terms of service', 'machine learning', 'artificial intelligence', 'generative ai', 'silicon valley', 'google cloud', 'amazon web', 'data science', 'operating system', 'software engineer', 'product manager', 'project manager', 'data analyst', 'gross margin', 'revenue growth', 'source code', 'version control', 'large language model'
|
|
459
|
+
, 'quarterly results', 'strategic planning', 'market research', 'customer base', 'privacy settings', 'account settings', 'security settings', 'download now', 'free trial', 'limited time', 'copyright protected', 'all rights', 'rights reserved', 'credit score', 'monthly rent', 'lease application', 'property address', 'reference number', 'additional identifier', 'lease agreement', 'hiring review', 'candidate name', 'privacy policy', 'terms of service', 'machine learning', 'artificial intelligence', 'generative ai', 'silicon valley', 'google cloud', 'amazon web', 'data science', 'operating system', 'software engineer', 'product manager', 'project manager', 'data analyst', 'gross margin', 'revenue growth', 'source code', 'version control', 'large language model',
|
|
460
|
+
'wages', 'wage', 'tips', 'compensation', 'withheld', 'withholding', 'medicare', 'deductions', 'deduction',
|
|
461
|
+
'regular', 'hours', 'holiday', 'overtime', 'commission', 'bonus', 'bonuses', 'records', 'record', 'statement', 'statements', 'rate', 'rates', 'current', 'ytd', 'benefits', 'taxable', 'pre-tax', 'post-tax', 'reimbursements', 'reimbursement', 'fica', 'oasdi', 'disability', 'unemployment', 'sui', 'sdi', 'std', 'ltd', 'exemptions', 'exemption', 'allowances', 'allowance', 'filing', 'status', 'single', 'married', 'head', 'household', 'advice', 'frequency', 'bi-weekly', 'biweekly', 'weekly', 'monthly', 'semi-monthly', 'direct', 'deposit', 'routing', 'box', 'boxes', 'code', 'control', 'omb', 'copy', 'instructions', 'information', 'deferred', 'adoption', 'statutory', 'third-party', 'sick', 'form', 'schedule', 'w-2', 'w2', 'w-4', 'w4', '1099', 'k-1', '1040', 'fed', 'med', 'fwt', 'swt', 'fed w/h', 'fed med', 'locality', 'state wages', 'state tax', 'local wages', 'local tax', 'allocated', 'nonqualified', 'suff', 'suffix', 'allocated tips', 'advance eic', 'advance eic payment', 'dependent care', 'dependent care benefits', 'nonqualified plans', 'statutory employee', 'retirement plan', 'third-party sick pay']);
|
|
378
462
|
|
|
379
|
-
let JARGON_WORDS = new Set(['step', 'page', 'grade', 'version', 'course', 'class', 'follow', 'chapter', 'lesson', 'unit', 'marketing', 'manager', 'specialist', 'science', 'administration', 'university', 'skills', 'leadership', 'communication', 'working', 'proficiency', 'decision', 'driven', 'experience', 'summary', 'bachelor', 'ads', 'solutions', 'positioning', 'acquisition', 'strategy', 'research', 'database', 'forecast', 'interest', 'prepared', 'merchant', 'document', 'feedback', 'template', 'campaign', 'partners', 'settings', 'keystone', 'llm', 'gpt', 'chatgpt', 'openai', 'anthropic', 'claude', 'gemini', 'api', 'json', 'xml', 'html', 'css', 'javascript', 'python', 'policy', 'terms', 'conditions', 'release', 'sprint', 'deployment', 'cluster', 'instance', 'package', 'module', 'revenue', 'margin', 'gross', 'quarter', 'system', 'code', 'data', 'cloud', 'server', 'database', 'artificial', 'intelligence', 'learning', 'generative']);
|
|
463
|
+
let JARGON_WORDS = new Set(['step', 'page', 'grade', 'version', 'course', 'class', 'follow', 'chapter', 'lesson', 'unit', 'marketing', 'manager', 'specialist', 'science', 'administration', 'university', 'skills', 'leadership', 'communication', 'working', 'proficiency', 'decision', 'driven', 'experience', 'summary', 'bachelor', 'ads', 'solutions', 'positioning', 'acquisition', 'strategy', 'research', 'database', 'forecast', 'interest', 'prepared', 'merchant', 'document', 'feedback', 'template', 'campaign', 'partners', 'settings', 'keystone', 'llm', 'gpt', 'chatgpt', 'openai', 'anthropic', 'claude', 'gemini', 'api', 'json', 'xml', 'html', 'css', 'javascript', 'python', 'golang', 'typescript', 'rust', 'fastapi', 'snowflake', 'kubernetes', 'terraform', 'docker', 'redis', 'kafka', 'pytorch', 'policy', 'terms', 'conditions', 'release', 'sprint', 'deployment', 'cluster', 'instance', 'package', 'module', 'revenue', 'margin', 'gross', 'quarter', 'system', 'code', 'data', 'cloud', 'server', 'database', 'artificial', 'intelligence', 'learning', 'generative', 'regular', 'hours', 'holiday', 'earnings', 'deductions', 'withheld', 'withholding', 'taxes', 'medicare', 'benefits', 'reimbursements', 'compensation', 'wages', 'code']);
|
|
380
464
|
|
|
381
465
|
let NOT_NAME_WORDS = new Set([
|
|
382
466
|
// Grammatical & Sentence Starters
|
|
383
|
-
'the', 'a', 'an', 'this', 'that', 'these', 'those', 'my', 'your', 'his', 'her', 'their', 'our', 'its', 'it', 'he', 'she', 'they', 'we', 'i', 'you', 'who', 'whom', 'which', 'what', 'whose', 'why', 'how', 'when', 'where', 'with', 'for', 'from', 'by', 'to', 'at', 'in', 'on', 'of', 'about', 'as', 'into', 'through', 'during', 'before', 'after', 'above', 'below', 'and', 'but', 'or', 'so', 'yet',
|
|
467
|
+
'the', 'a', 'an', 'this', 'that', 'these', 'those', 'my', 'your', 'his', 'her', 'their', 'our', 'its', 'it', 'he', 'she', 'they', 'we', 'i', 'you', 'who', 'whom', 'which', 'what', 'whose', 'why', 'how', 'when', 'where', 'with', 'for', 'from', 'by', 'to', 'at', 'in', 'on', 'of', 'about', 'as', 'into', 'through', 'during', 'before', 'after', 'above', 'below', 'and', 'but', 'or', 'so', 'yet', 'im', "i'm", "you're", "they're", "we're", "it's", "he's", "she's", "that's", "there's", "what's", "who's", "i've", "you've", "we've", "they've", "i'll", "you'll", "we'll", "they'll", "i'd", "you'd", "we'd", "they'd",
|
|
468
|
+
// Verbs, Auxiliaries, Commands & Imperatives
|
|
469
|
+
'do', 'does', 'did', 'done', 'doing', 'dont', "don't", 'doesnt', "doesn't", 'didnt', "didn't", 'not', 'no', 'never', 'always',
|
|
470
|
+
'be', 'is', 'am', 'are', 'was', 'were', 'been', 'being',
|
|
471
|
+
'have', 'has', 'had', 'having',
|
|
472
|
+
'can', 'could', 'may', 'might', 'must', 'shall', 'should', 'will', 'would', 'wont', "won't", 'wouldnt', "wouldn't", 'shouldnt', "shouldn't", 'couldnt', "couldn't", 'cant', "can't", 'cannot',
|
|
473
|
+
'write', 'writing', 'written', 'writes', 'read', 'reading', 'reads',
|
|
474
|
+
'wait', 'waiting', 'waited', 'waits', 'place', 'placing', 'placed', 'places',
|
|
475
|
+
'display', 'displaying', 'displayed', 'displays',
|
|
476
|
+
'provide', 'providing', 'provided', 'provides',
|
|
477
|
+
'show', 'showing', 'shown', 'shows',
|
|
478
|
+
'tell', 'telling', 'told', 'tells',
|
|
479
|
+
'ask', 'asking', 'asked', 'asks',
|
|
480
|
+
'use', 'using', 'used', 'uses',
|
|
481
|
+
'select', 'selecting', 'selected', 'selects',
|
|
482
|
+
'find', 'finding', 'findings', 'found', 'finds',
|
|
483
|
+
'reference', 'referencing', 'referenced', 'references',
|
|
484
|
+
'access', 'accessing', 'accessed', 'accesses',
|
|
485
|
+
'note', 'noting', 'noted', 'notes',
|
|
486
|
+
'get', 'getting', 'got', 'gotten', 'gets',
|
|
487
|
+
'make', 'making', 'made', 'makes',
|
|
488
|
+
'give', 'giving', 'given', 'gives',
|
|
489
|
+
'take', 'taking', 'took', 'taken', 'takes',
|
|
490
|
+
'put', 'putting', 'puts',
|
|
491
|
+
'set', 'setting', 'sets',
|
|
492
|
+
'keep', 'keeping', 'kept', 'keeps',
|
|
493
|
+
'let', 'letting', 'lets',
|
|
494
|
+
'leave', 'leaving', 'left', 'leaves',
|
|
495
|
+
'run', 'running', 'ran', 'runs',
|
|
496
|
+
'stop', 'stopping', 'stopped', 'stops',
|
|
497
|
+
'start', 'starting', 'started', 'starts',
|
|
498
|
+
'check', 'checking', 'checked', 'checks',
|
|
499
|
+
'print', 'printing', 'printed', 'prints',
|
|
500
|
+
'generate', 'generating', 'generated', 'generates',
|
|
501
|
+
'create', 'creating', 'created', 'creates',
|
|
502
|
+
'build', 'building', 'built', 'builds',
|
|
503
|
+
'include', 'including', 'included', 'includes',
|
|
504
|
+
'exclude', 'excluding', 'excluded', 'excludes',
|
|
505
|
+
'format', 'formatting', 'formatted', 'formats',
|
|
506
|
+
'change', 'changing', 'changed', 'changes',
|
|
507
|
+
'send', 'sending', 'sent', 'sends',
|
|
508
|
+
'receive', 'receiving', 'received', 'receives',
|
|
509
|
+
'delete', 'deleting', 'deleted', 'deletes',
|
|
510
|
+
'remove', 'removing', 'removed', 'removes',
|
|
511
|
+
'insert', 'inserting', 'inserted', 'inserts',
|
|
512
|
+
'update', 'updating', 'updated', 'updates',
|
|
513
|
+
'review', 'reviewing', 'reviewed', 'reviews',
|
|
514
|
+
'allow', 'allowing', 'allowed', 'allows',
|
|
515
|
+
'deny', 'denying', 'denied', 'denies',
|
|
516
|
+
'require', 'requiring', 'required', 'requires',
|
|
517
|
+
'turn', 'turning', 'turned', 'turns',
|
|
518
|
+
'switch', 'switching', 'switched', 'switches',
|
|
519
|
+
'enable', 'enabling', 'enabled', 'enables',
|
|
520
|
+
'disable', 'disabling', 'disabled', 'disables',
|
|
521
|
+
'ensure', 'ensuring', 'ensured', 'ensures',
|
|
522
|
+
'verify', 'verifying', 'verified', 'verifies',
|
|
523
|
+
'execute', 'executing', 'executed', 'executes',
|
|
524
|
+
'test', 'testing', 'tested', 'tests',
|
|
525
|
+
'install', 'installing', 'installed', 'installs',
|
|
526
|
+
'uninstall', 'uninstalling', 'uninstalled', 'uninstalls',
|
|
527
|
+
'suppose', 'supposed', 'supposing', 'supposes',
|
|
528
|
+
'respond', 'responding', 'responded', 'responds',
|
|
529
|
+
'preserve', 'preserving', 'preserved', 'preserves',
|
|
530
|
+
'replace', 'replacing', 'replaced', 'replaces',
|
|
531
|
+
'present', 'presented', 'presenting', 'presents',
|
|
532
|
+
'admit', 'admitted', 'admitting', 'admits',
|
|
533
|
+
'complain', 'complained', 'complaining', 'complains',
|
|
534
|
+
'prescribe', 'prescribed', 'prescribing', 'prescribes',
|
|
535
|
+
'diagnose', 'diagnosed', 'diagnosing', 'diagnoses',
|
|
536
|
+
'report', 'reported', 'reporting', 'reports',
|
|
537
|
+
'state', 'stated', 'stating', 'states',
|
|
538
|
+
'undergo', 'underwent', 'undergoing', 'undergoes',
|
|
539
|
+
'experience', 'experienced', 'experiencing', 'experiences',
|
|
540
|
+
'arrive', 'arrived', 'arriving', 'arrives',
|
|
541
|
+
'order', 'ordered', 'ordering', 'orders',
|
|
542
|
+
// Adverbs, Prepositions, Conjunctions & Modifiers
|
|
543
|
+
'again', 'without', 'with', 'within', 'specific', 'specifically', 'permission', 'permissions',
|
|
544
|
+
'underneath', 'above', 'below', 'between', 'among', 'together', 'separately', 'instead',
|
|
545
|
+
'also', 'too', 'either', 'neither', 'both', 'each', 'every', 'all', 'some', 'any', 'none',
|
|
546
|
+
'only', 'just', 'already', 'currently', 'more', 'most', 'less', 'least',
|
|
547
|
+
'very', 'quite', 'rather', 'such', 'same', 'different', 'other', 'others', 'another',
|
|
548
|
+
'like', 'unlike', 'similar', 'complete', 'completely', 'entire', 'entirely',
|
|
549
|
+
'exact', 'exactly', 'approximate', 'approximately', 'general', 'generally',
|
|
550
|
+
'direct', 'directly', 'indirect', 'indirectly', 'total', 'totally', 'full', 'fully',
|
|
551
|
+
'partial', 'partially', 'own', 'proper', 'properly',
|
|
552
|
+
'now', 'then', 'soon', 'later', 'here', 'there', 'everywhere', 'nowhere', 'somewhere', 'anywhere',
|
|
553
|
+
'inside', 'outside', 'before', 'after', 'since', 'until', 'till',
|
|
554
|
+
'while', 'whereas', 'unless', 'although', 'though', 'even', 'because',
|
|
555
|
+
'therefore', 'however', 'furthermore', 'moreover', 'meanwhile', 'otherwise', 'besides', 'further',
|
|
384
556
|
// Greetings & Salutations
|
|
385
557
|
'hello', 'hi', 'hey', 'dear', 'greetings',
|
|
386
558
|
// Document & Resume Structure
|
|
387
|
-
'summary', 'experience', 'education', 'skills', 'languages', 'project', 'history', 'background', 'objective', 'profile', 'awards', 'honors', 'certifications', 'publications', 'interests', 'references',
|
|
559
|
+
'summary', 'experience', 'education', 'skills', 'languages', 'project', 'history', 'background', 'objective', 'profile', 'awards', 'honors', 'certifications', 'publications', 'interests', 'references', 'statement', 'statements', 'form', 'forms',
|
|
388
560
|
// Business & Job Roles
|
|
389
561
|
'manager', 'director', 'specialist', 'analyst', 'engineer', 'developer', 'consultant', 'officer', 'representative', 'agent', 'lead', 'leader', 'president', 'coordinator', 'admin', 'administrator', 'executive', 'founder', 'partner', 'intern', 'trainee', 'advisor', 'head', 'vp', 'chief',
|
|
390
562
|
// Departments & Fields
|
|
391
563
|
'marketing', 'sales', 'engineering', 'finance', 'accounting', 'legal', 'operations', 'support', 'recruiting', 'talent', 'acquisition', 'compliance', 'security', 'technical', 'development', 'product', 'design', 'creative', 'strategy', 'planning', 'analytics', 'science', 'business', 'administration',
|
|
392
564
|
// Tools & Tech Concepts
|
|
393
|
-
'google', 'ads', 'analytics', 'meta', 'hubspot', 'crm', 'salesforce', 'wordpress', 'mailchimp', 'adobe', 'figma', 'canva', 'slack', 'zoom', 'teams', 'microsoft', 'office', 'excel', 'word', 'powerpoint', 'notion', 'jira', 'confluence', 'github', 'gitlab', 'aws', 'azure', 'cloud', 'database', 'sql', 'python', 'java', 'javascript', 'html', 'css', 'react', 'node', 'api', 'saas', 'b2b', 'b2c', 'url', 'domain', 'website', 'app', 'application', 'software', 'email', 'phone', 'contact', 'address',
|
|
565
|
+
'google', 'ads', 'analytics', 'meta', 'hubspot', 'crm', 'salesforce', 'wordpress', 'mailchimp', 'adobe', 'figma', 'canva', 'slack', 'zoom', 'teams', 'microsoft', 'office', 'excel', 'word', 'powerpoint', 'notion', 'jira', 'confluence', 'github', 'gitlab', 'aws', 'gcp', 'azure', 'cloud', 'database', 'sql', 'python', 'golang', 'typescript', 'rust', 'fastapi', 'snowflake', 'kubernetes', 'terraform', 'docker', 'redis', 'kafka', 'pytorch', 'java', 'javascript', 'html', 'css', 'react', 'node', 'api', 'saas', 'b2b', 'b2c', 'url', 'domain', 'website', 'app', 'application', 'software', 'email', 'phone', 'contact', 'address',
|
|
394
566
|
|
|
395
567
|
// General Academic & Professional vocabulary
|
|
396
568
|
'bachelor', 'master', 'doctor', 'associate', 'degree', 'university', 'college', 'school', 'institute', 'academy', 'graduated', 'major', 'minor', 'gpa', 'cum', 'laude', 'honors', 'deans', 'list', 'scholarship',
|
|
@@ -399,15 +571,29 @@ let NOT_NAME_WORDS = new Set([
|
|
|
399
571
|
// Common Resume / Business Phrases
|
|
400
572
|
'results-driven', 'data-driven', 'customer-centric', 'detail-oriented', 'cross-functional', 'self-motivated', 'time-management', 'problem-solving', 'fast-paced', 'year-over-year',
|
|
401
573
|
// Legal & Trust terms
|
|
402
|
-
'trust', 'trustee', 'co-trustee', 'settlor', 'grantor', 'beneficiary', 'agreement', 'will', 'estate', 'witness', 'declaration', 'signatory', 'testator', 'notary', 'commission', 'county', 'state', 'court', 'article', 'section', 'paragraph', 'schedule', 'exhibit', 'amendment', 'addendum', 'power', 'attorney', 'guardian', 'executor', 'administrator', 'survivor', 'predecessor', 'successor', 'whereof', 'hereby', 'thereby', 'herein', 'therein', 'witnesseth', 'whereas', 'therefore', 'now', 'dated', 'effective',
|
|
574
|
+
'trust', 'trustee', 'co-trustee', 'settlor', 'grantor', 'beneficiary', 'agreement', 'will', 'estate', 'witness', 'declaration', 'signatory', 'testator', 'notary', 'commission', 'county', 'state', 'court', 'article', 'section', 'paragraph', 'schedule', 'exhibit', 'amendment', 'addendum', 'power', 'attorney', 'guardian', 'executor', 'administrator', 'survivor', 'predecessor', 'successor', 'whereof', 'hereby', 'thereby', 'herein', 'therein', 'witnesseth', 'whereas', 'therefore', 'now', 'dated', 'effective', 'matter', 'case', 'cases', 'docket', 'dockets', 'number', 'numbers', 'firm', 'firms', 'lawyer', 'lawyers', 'counsel', 'counsels', 'counselstack', 'tier', 'tiers', 'finding', 'findings', 'connector', 'connectors', 'platform', 'platforms',
|
|
575
|
+
// Medical & Clinical terms
|
|
576
|
+
'clinical', 'note', 'notes', 'dx', 'rx', 'tx', 'hx', 'px', 'sx', 'type', 'diabetes', 'referred', 'referral', 'diagnosed', 'diagnosis', 'patient', 'insurance', 'bcbs', 'mrn', 'dob',
|
|
577
|
+
// Tax & Payroll terms
|
|
578
|
+
'wages', 'wage', 'tips', 'compensation', 'withheld', 'withholding', 'medicare', 'deductions', 'deduction', 'earning', 'earnings', 'gross', 'net', 'pay', 'payroll', 'paystub', 'taxable', 'exempt', 'allowance', 'allowances', 'regular', 'hours', 'holiday', 'overtime', 'commission', 'bonus', 'bonuses', 'records', 'record', 'statement', 'statements', 'rate', 'rates', 'current', 'ytd', 'benefits', 'taxable', 'pre-tax', 'post-tax', 'reimbursements', 'reimbursement', 'fica', 'oasdi', 'disability', 'unemployment', 'sui', 'sdi', 'std', 'ltd', 'exemptions', 'exemption', 'allowances', 'allowance', 'filing', 'status', 'single', 'married', 'head', 'household', 'advice', 'frequency', 'bi-weekly', 'biweekly', 'weekly', 'monthly', 'semi-monthly', 'direct', 'deposit', 'routing', 'box', 'boxes', 'code', 'control', 'omb', 'copy', 'instructions', 'information', 'deferred', 'adoption', 'statutory', 'third-party', 'sick', 'form', 'schedule', 'w-2', 'w2', 'w-4', 'w4', '1099', 'k-1', '1040', 'fed', 'med', 'fwt', 'swt', 'fed w/h', 'fed med', 'locality', 'state wages', 'state tax', 'local wages', 'local tax', 'allocated', 'nonqualified',
|
|
579
|
+
// Common Web, UI, Compliance, Document & AI Terms (Suppresses false-positive Name detection on headlines, buttons, and badges)
|
|
580
|
+
'incident', 'incidents', 'critical', 'production', 'impacted', 'reported', 'details', 'vulnerability', 'vulnerabilities', 'host', 'types', 'type', 'leaked', 'leak', 'leaks', 'masked', 'mask', 'masking', 'leave', 'screen', 'screens', 'risk', 'risks', 'cluster', 'clusters', 'parameter', 'parameters', 'processing', 'process', 'processed', 'verified', 'verify', 'verification', 'playground', 'guide', 'guides', 'protection', 'protect', 'corporate', 'enterprise', 'log', 'logs', 'airplane', 'mode', 'zero', 'trust', 'top', 'data', 'live', 'scrubber', 'scrub', 'scrubbed', 'note', 'notes', 'secret', 'secrets', 'card', 'cards', 'raw', 'input', 'output', 'contains', 'contain', 'contained', 'platform', 'solutions', 'pricing', 'company', 'news', 'dashboard', 'add', 'chrome', 'sample', 'samples', 'try', 'terms', 'privacy', 'policy', 'policies', 'home', 'compliance', 'framework', 'frameworks', 'audit', 'audits', 'receipt', 'receipts', 'overview', 'explore', 'vectors', 'vector', 'standard', 'standards', 'status', 'preview', 'view', 'actions', 'action', 'button', 'buttons', 'option', 'options', 'general', 'specialized', 'custom', 'rule', 'rules', 'token', 'tokens', 'value', 'values', 'session', 'sessions', 'local', 'server', 'servers', 'cloud', 'ram', 'memory', 'offline', 'online', 'client', 'browser', 'extension', 'workspace', 'workplace', 'pan', 'phi', 'pii', 'soc', 'soc2', 'gdpr', 'hipaa', 'ccpa', 'iso27001', 'pci', 'dss', 'nist', 'chatgpt', 'claude', 'gemini', 'copilot', 'perplexity', 'deepseek', 'qwen', 'grok', 'llama', 'mistral', 'ai', 'llm', 'prompt', 'prompts', 'transmission', 'transit', 'egress', 'neutralized', 'stripped', 'isolated', 'isolation', 'unlocked', 'locked', 'unlock', 'download', 'copy', 'dismiss', 'close', 'save', 'settings', 'protect', 'reveal', 'unmask', 'restore', 'restored', 'export', 'import',
|
|
581
|
+
// Games, Chess, and Playing Pieces
|
|
582
|
+
'bishop', 'bishops', 'knight', 'knights', 'rook', 'rooks', 'pawn', 'pawns', 'king', 'kings', 'queen', 'queens', 'chessboard', 'checkmate', 'stalemate', 'castling', 'en passant', 'chess',
|
|
583
|
+
// Colors & Visual Descriptors
|
|
584
|
+
'white', 'black', 'red', 'blue', 'green', 'yellow', 'orange', 'purple', 'pink', 'brown', 'gray', 'grey', 'dark', 'light', 'gold', 'silver', 'bronze',
|
|
585
|
+
// Animation, 3D Rendering & Prompt Terminology
|
|
586
|
+
'pixar', 'disney', 'animation', 'render', 'rendering', 'composition', 'cinematic', 'smooth', 'glides', 'glide', 'gliding', 'capture', 'captures', 'capturing', 'camera', 'orbit', 'orbits', 'orbiting', 'trapped', 'trap', 'trapping', 'square', 'squares', 'character', 'characters', 'expressive', 'living', 'texture', 'textures', 'reflection', 'reflections', 'grain', 'candlelight', 'wooden', 'polished', 'vertical', 'horizontal', 'macro', 'closeup', 'close-up', 'scene', 'scenes', 'shot', 'shots', 'shadow', 'shadows',
|
|
587
|
+
// Email, Outreach, Guest Posting & Agency Business Vocabulary
|
|
588
|
+
'guest', 'post', 'posts', 'posting', 'attached', 'attach', 'attachment', 'attachments', 'updated', 'update', 'updates', 'list', 'lists', 'line', 'lines', 'rate', 'rates', 'affordable', 'services', 'service', 'infotech', 'technologies', 'technology', 'agency', 'agencies', 'digital', 'marketing', 'traffic', 'smart', 'design', 'seo', 'per', 'host', 'hosting', 'sites', 'site', 'inbox', 'starred', 'snoozed', 'important', 'sent', 'drafts', 'draft', 'spam', 'bin', 'trash', 'purchases', 'travel', 'social', 'forums', 'promotions', 'promotion', 'reply', 'forward', 'labels', 'label', 'compose', 'message', 'messages', 'mailer', 'outreach', 'backlink', 'backlinks', 'domain', 'authority', 'da', 'dr', 'founder', 'ceo', 'cto', 'cfo', 'coo', 'vp', 'head', 'lead',
|
|
403
589
|
// US States
|
|
404
|
-
'california', 'texas', 'florida', 'york', 'illinois', 'pennsylvania', 'ohio', 'georgia', 'michigan', 'carolina', 'virginia', 'washington', 'arizona', 'massachusetts', 'tennessee', 'indiana', 'maryland', 'missouri', 'wisconsin', 'colorado', 'minnesota', 'alabama', 'louisiana', 'kentucky', 'oregon', 'oklahoma', 'connecticut', 'utah', 'iowa', 'nevada', 'arkansas', 'mississippi', 'kansas', 'new mexico', 'nebraska', 'idaho', 'hawaii', 'maine', 'new hampshire', 'rhode island', 'montana', 'delaware', 'south dakota', 'north dakota', 'alaska', 'vermont', 'wyoming'
|
|
405
|
-
|
|
590
|
+
'california', 'texas', 'florida', 'york', 'illinois', 'pennsylvania', 'ohio', 'georgia', 'michigan', 'carolina', 'virginia', 'washington', 'arizona', 'massachusetts', 'tennessee', 'indiana', 'maryland', 'missouri', 'wisconsin', 'colorado', 'minnesota', 'alabama', 'louisiana', 'kentucky', 'oregon', 'oklahoma', 'connecticut', 'utah', 'iowa', 'nevada', 'arkansas', 'mississippi', 'kansas', 'new mexico', 'nebraska', 'idaho', 'hawaii', 'maine', 'new hampshire', 'rhode island', 'montana', 'delaware', 'south dakota', 'north dakota', 'alaska', 'vermont', 'wyoming',
|
|
591
|
+
'f.3d', 'f.supp', 'u.s.c.', 'v.', 'plaintiff', 'defendant', 'v', 'u.s.', 'court', 'app.', 'reporter', 'cir.']);
|
|
406
592
|
|
|
407
593
|
const PROFILE_JARGON = {
|
|
408
|
-
medical: ['sleep', 'apnea', 'symptom', 'symptoms', 'trauma', 'hypertension', 'health', 'disease', 'condition', 'diagnosis', 'treatment', 'medication', 'dose', 'patient', 'clinic', 'surgery', 'therapy', 'alcohol', 'cannabis', 'blood', 'pressure', 'heart', 'rate', 'emergency', 'contact', 'relationship'],
|
|
594
|
+
medical: ['sleep', 'apnea', 'symptom', 'symptoms', 'trauma', 'hypertension', 'health', 'disease', 'condition', 'diagnosis', 'treatment', 'medication', 'dose', 'patient', 'clinic', 'surgery', 'therapy', 'alcohol', 'cannabis', 'blood', 'pressure', 'heart', 'rate', 'emergency', 'contact', 'relationship', 'type', 'diabetes', 'cancer', 'asthma', 'copd', 'covid', 'infection', 'syndrome', 'disorder', 'chronic', 'acute', 'illness', 'fever', 'allergy', 'pain', 'referral', 'referred', 'prescription', 'prescribed', 'doctor', 'physician', 'nurse', 'hospital', 'clinical', 'note', 'notes', 'dx', 'rx', 'tx', 'hx', 'px', 'sx', 'insurance', 'bcbs', 'bronchitis', 'amoxicillin', 'penicillin', 'antibiotic', 'antibiotics', 'vital', 'vitals', 'bp', 'hr', 'bpm', 'mmhg', 'allergies', 'dosage'],
|
|
409
595
|
realestate: ['escrow', 'tenant', 'landlord', 'lease', 'mortgage', 'appraisal', 'broker', 'property', 'zoning', 'parcel', 'rent', 'buyer', 'seller', 'agent', 'listing'],
|
|
410
|
-
legal: ['testator', 'notary', 'commission', 'county', 'court', 'affidavit', 'plaintiff', 'defendant', 'litigation', 'jurisdiction', 'agreement', 'contract', 'settlement', 'clause', 'article', 'section'],
|
|
596
|
+
legal: ['testator', 'notary', 'commission', 'county', 'court', 'affidavit', 'plaintiff', 'defendant', 'litigation', 'jurisdiction', 'agreement', 'contract', 'settlement', 'clause', 'article', 'section', 'matter', 'case'],
|
|
411
597
|
hr: ['candidate', 'employee', 'payroll', 'benefits', 'salary', 'vacation', 'supervisor', 'subordinate', 'performance', 'appraisal', 'interview', 'resume', 'applicant'],
|
|
412
598
|
sales: ['prospect', 'opportunity', 'quota', 'pipeline', 'deal', 'revenue', 'forecast', 'lead', 'churn', 'client', 'customer']
|
|
413
599
|
};
|
|
@@ -447,11 +633,14 @@ const PROFILE_JARGON = {
|
|
|
447
633
|
}
|
|
448
634
|
|
|
449
635
|
const PROFILE_ALIAS_MAP = {
|
|
636
|
+
'general': 'general',
|
|
637
|
+
'underwriting': 'underwriting', 'lending': 'underwriting', 'mortgage': 'underwriting', 'loan': 'underwriting', 'income': 'underwriting', 'income_verification': 'underwriting', 'payroll': 'underwriting', 'w2': 'underwriting', 'paystub': 'underwriting',
|
|
450
638
|
'medical': 'medical', 'healthcare': 'medical', 'health': 'medical', 'pharma': 'pharma',
|
|
451
|
-
'engineering': 'engineering', 'dev': 'engineering', 'tech': 'tech',
|
|
452
|
-
'finance': 'finance', 'bizops': 'bizops', 'sales': 'sales', 'wealthmgmt': 'wealthmgmt', 'insurance': 'insurance', 'accounting': 'accounting',
|
|
639
|
+
'engineering': 'engineering', 'dev': 'engineering', 'devops': 'engineering', 'tech': 'tech',
|
|
640
|
+
'finance': 'finance', 'bizops': 'bizops', 'sales': 'sales', 'wealthmgmt': 'wealthmgmt', 'wealth': 'wealthmgmt', 'insurance': 'insurance', 'accounting': 'accounting',
|
|
453
641
|
'legal': 'legal', 'compliance': 'compliance', 'ccpa': 'ccpa',
|
|
454
642
|
'hr': 'hr', 'security': 'security', 'marketing': 'marketing', 'support': 'support',
|
|
643
|
+
'realestate': 'realestate', 'academic': 'academic', 'agents': 'agents', 'ai_agents': 'agents', 'creative': 'creative', 'personal': 'personal'
|
|
455
644
|
};
|
|
456
645
|
|
|
457
646
|
function getActiveRules(activeProfile) {
|
|
@@ -459,7 +648,7 @@ const PROFILE_JARGON = {
|
|
|
459
648
|
if (activeProfile && activeProfile.toLowerCase() !== 'general') {
|
|
460
649
|
const canonicalProfile = PROFILE_ALIAS_MAP[activeProfile.toLowerCase()] || 'general';
|
|
461
650
|
if (canonicalProfile !== 'general' && PROFILE_RULES[canonicalProfile]) {
|
|
462
|
-
activeRules =
|
|
651
|
+
activeRules = PROFILE_RULES[canonicalProfile].concat(activeRules);
|
|
463
652
|
}
|
|
464
653
|
}
|
|
465
654
|
return activeRules;
|
|
@@ -482,14 +671,15 @@ const PROFILE_JARGON = {
|
|
|
482
671
|
|
|
483
672
|
if (customRules && customRules.length > 0) {
|
|
484
673
|
const sorted = [...customRules].sort((a, b) => {
|
|
485
|
-
const patternA = typeof a === 'string' ? a : a.pattern;
|
|
486
|
-
const patternB = typeof b === 'string' ? b : b.pattern;
|
|
674
|
+
const patternA = typeof a === 'string' ? a : (a.pattern || (a.regex ? a.regex.source : '') || '');
|
|
675
|
+
const patternB = typeof b === 'string' ? b : (b.pattern || (b.regex ? b.regex.source : '') || '');
|
|
487
676
|
return patternB.length - patternA.length;
|
|
488
677
|
});
|
|
489
678
|
|
|
490
679
|
sorted.forEach(cr => {
|
|
491
|
-
const pattern = typeof cr === 'string' ? cr : cr.pattern;
|
|
492
|
-
|
|
680
|
+
const pattern = typeof cr === 'string' ? cr : (cr.pattern || (cr.regex ? cr.regex.source : ''));
|
|
681
|
+
if (!pattern) return;
|
|
682
|
+
const label = typeof cr === 'string' ? 'CUSTOM' : (cr.label || cr.mask || cr.name || 'CUSTOM');
|
|
493
683
|
|
|
494
684
|
let rx;
|
|
495
685
|
try {
|
|
@@ -525,33 +715,56 @@ const PROFILE_JARGON = {
|
|
|
525
715
|
let matchedText = m[0];
|
|
526
716
|
let start = m.index;
|
|
527
717
|
|
|
528
|
-
if (
|
|
718
|
+
if (m.length > 1 && m[1] !== undefined && m[1] !== '') {
|
|
529
719
|
matchedText = m[1];
|
|
530
|
-
|
|
720
|
+
const relOffset = m[0].indexOf(m[1]);
|
|
721
|
+
if (relOffset !== -1) {
|
|
722
|
+
start = m.index + relOffset;
|
|
723
|
+
}
|
|
531
724
|
}
|
|
532
725
|
const end = start + matchedText.length;
|
|
533
726
|
|
|
534
727
|
if (rule.type !== 'NAME' && rule.type !== 'ADDRESS') {
|
|
728
|
+
const val = matchedText.toLowerCase().trim();
|
|
729
|
+
if (NAME_STOP_LIST.has(val) || NOT_NAME_WORDS.has(val)) {
|
|
730
|
+
// Skip if generic English dictionary term matched by greedy regex (e.g. SWIFT matching SCRUBBER or CONTAINS)
|
|
731
|
+
if (rule.type === 'FINANCIAL' || rule.type === 'ID' || rule.type === 'PRIVACY' || rule.type === 'SECRET') {
|
|
732
|
+
continue;
|
|
733
|
+
}
|
|
734
|
+
}
|
|
535
735
|
matches.push({ start, end, value: matchedText, type: rule.type });
|
|
536
736
|
} else {
|
|
737
|
+
if (rule.type === 'NAME') {
|
|
738
|
+
matchedText = matchedText.replace(/[.,;:]+$/, '').trim();
|
|
739
|
+
}
|
|
537
740
|
let val = matchedText.toLowerCase().trim();
|
|
538
|
-
if (NAME_STOP_LIST.has(val)) continue;
|
|
741
|
+
if (!val || NAME_STOP_LIST.has(val)) continue;
|
|
539
742
|
|
|
540
|
-
if (
|
|
743
|
+
if (rule.type === 'NAME') {
|
|
541
744
|
let words = val.split(/[ \t\xA0]+/);
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
745
|
+
if (!rule.isContextName) {
|
|
746
|
+
let origWords = matchedText.split(/[ \t\xA0]+/);
|
|
747
|
+
while (words.length > 2 && (currentJargon.has(words[0]) || (words[0].length > 1 && NOT_NAME_WORDS.has(words[0])) || (words[0].replace(/[^\p{L}]/gu, '').length > 1 && NOT_NAME_WORDS.has(words[0].replace(/[^\p{L}]/gu, ''))))) {
|
|
748
|
+
origWords.shift();
|
|
749
|
+
words.shift();
|
|
750
|
+
const nextStart = matchedText.indexOf(origWords[0]);
|
|
751
|
+
if (nextStart !== -1) {
|
|
752
|
+
start += nextStart;
|
|
753
|
+
matchedText = matchedText.substring(nextStart);
|
|
754
|
+
val = matchedText.toLowerCase().trim();
|
|
755
|
+
} else {
|
|
756
|
+
break;
|
|
757
|
+
}
|
|
758
|
+
}
|
|
550
759
|
}
|
|
551
|
-
if (words.some(w =>
|
|
760
|
+
if (!rule.isContextName && words.some(w => {
|
|
761
|
+
const cleanW = w.replace(/[^\p{L}]/gu, '');
|
|
762
|
+
if (cleanW.length <= 1) return false;
|
|
763
|
+
return currentJargon.has(w) || NOT_NAME_WORDS.has(w) || NOT_NAME_WORDS.has(cleanW) || NAME_STOP_LIST.has(w) || NAME_STOP_LIST.has(cleanW);
|
|
764
|
+
})) continue;
|
|
552
765
|
} else if (rule.type === 'ADDRESS') {
|
|
553
|
-
const
|
|
554
|
-
if (
|
|
766
|
+
const cleanVal = val.replace(/[.,;!?]/g, ' ').trim();
|
|
767
|
+
if (NAME_STOP_LIST.has(cleanVal) || currentJargon.has(cleanVal)) continue;
|
|
555
768
|
}
|
|
556
769
|
|
|
557
770
|
matches.push({ start, end: start + matchedText.length, value: matchedText, type: rule.type });
|
|
@@ -588,7 +801,7 @@ const PROFILE_JARGON = {
|
|
|
588
801
|
nameWords.forEach(w => {
|
|
589
802
|
if (w && w.length >= 2 && /^\p{Lu}/u.test(w)) {
|
|
590
803
|
const wl = w.toLowerCase();
|
|
591
|
-
if (!NOT_NAME_WORDS.has(wl) && !JARGON_WORDS.has(wl) && !NAME_STOP_LIST.has(wl)) {
|
|
804
|
+
if (!NOT_NAME_WORDS.has(wl) && !JARGON_WORDS.has(wl) && !NAME_STOP_LIST.has(wl) && !currentJargon.has(wl)) {
|
|
592
805
|
learnedNames.add(w);
|
|
593
806
|
}
|
|
594
807
|
}
|
|
@@ -641,13 +854,24 @@ const PROFILE_JARGON = {
|
|
|
641
854
|
return Array.from(aliases);
|
|
642
855
|
}
|
|
643
856
|
|
|
857
|
+
function formatToken(label, index, format = 'brackets') {
|
|
858
|
+
const cleanLabel = String(label || 'PII').replace(/[^A-Za-z0-9_]/g, '_').toUpperCase();
|
|
859
|
+
switch(format) {
|
|
860
|
+
case 'xml': return `<${cleanLabel}_${index}>`;
|
|
861
|
+
case 'mustache': return `{{${cleanLabel}_${index}}}`;
|
|
862
|
+
case 'underscores': return `__${cleanLabel}_${index}__`;
|
|
863
|
+
case 'brackets':
|
|
864
|
+
default: return `[${cleanLabel}_${index}]`;
|
|
865
|
+
}
|
|
866
|
+
}
|
|
867
|
+
|
|
644
868
|
function buildRestorationRegexAndRules(tokenMap) {
|
|
645
869
|
const ObjectKeys = Object.keys(tokenMap);
|
|
646
870
|
if (ObjectKeys.length === 0) return { compositeRegex: null, looseRules: [] };
|
|
647
871
|
|
|
648
872
|
const sortedKeys = [...ObjectKeys].sort((a, b) => {
|
|
649
|
-
const innerA = a.replace(/^\[|\]
|
|
650
|
-
const innerB = b.replace(/^\[|\]
|
|
873
|
+
const innerA = a.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
|
|
874
|
+
const innerB = b.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
|
|
651
875
|
const matchA = innerA.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
|
|
652
876
|
const matchB = innerB.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
|
|
653
877
|
if (matchA && matchB) {
|
|
@@ -665,7 +889,7 @@ const PROFILE_JARGON = {
|
|
|
665
889
|
const regexParts = [];
|
|
666
890
|
|
|
667
891
|
sortedKeys.forEach(k => {
|
|
668
|
-
const inner = k.replace(/^\[|\]
|
|
892
|
+
const inner = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
|
|
669
893
|
const match = inner.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
|
|
670
894
|
if (match) {
|
|
671
895
|
const label = match[1];
|
|
@@ -673,7 +897,8 @@ const PROFILE_JARGON = {
|
|
|
673
897
|
const aliases = getLabelAliases(label);
|
|
674
898
|
|
|
675
899
|
const escapedAliases = aliases.map(a => a.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
|
676
|
-
const
|
|
900
|
+
const aliasesGroup = `(?:${escapedAliases.join('|')})`;
|
|
901
|
+
const loosePattern = `(?:\\[\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*\\]|<\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*>|\\{\\{\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*\\}\\}|__\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*__|(?<![A-Za-z0-9\\u0400-\\u04FF_])${aliasesGroup}[-_\\s]*0*${baseIndex}(?![A-Za-z0-9\\u0400-\\u04FF_]))(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})?`;
|
|
677
902
|
looseRules.push({ token: k, pattern: loosePattern });
|
|
678
903
|
}
|
|
679
904
|
regexParts.push(k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
|
@@ -681,42 +906,157 @@ const PROFILE_JARGON = {
|
|
|
681
906
|
|
|
682
907
|
let compositeRegex = null;
|
|
683
908
|
if (regexParts.length > 0) {
|
|
684
|
-
compositeRegex = new RegExp(`(?:\\b|\\[)?(?:(?:(?<!\\w)|(?<=\\s))(?:${regexParts.join('|')})(?:(?!\\w)|(?=\\s)))(?:\\b|\\])?`, 'g');
|
|
909
|
+
compositeRegex = new RegExp(`(?:\\b|\\[|<|\\{\\{|__)?(?:(?:(?<!\\w)|(?<=\\s))(?:${regexParts.join('|')})(?:(?!\\w)|(?=\\s)))(?:\\b|\\]|>|\\}\\}|__)?(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})?`, 'g');
|
|
685
910
|
}
|
|
686
911
|
|
|
687
912
|
return { compositeRegex, looseRules };
|
|
688
913
|
}
|
|
689
914
|
|
|
915
|
+
function isJsonPayload(str) {
|
|
916
|
+
if (!str || typeof str !== "string") return false;
|
|
917
|
+
const trimmed = str.trim();
|
|
918
|
+
if (!((trimmed.startsWith("{") && trimmed.endsWith("}")) || (trimmed.startsWith("[") && trimmed.endsWith("]")))) {
|
|
919
|
+
return false;
|
|
920
|
+
}
|
|
921
|
+
try {
|
|
922
|
+
JSON.parse(trimmed);
|
|
923
|
+
return true;
|
|
924
|
+
} catch (_) {
|
|
925
|
+
return false;
|
|
926
|
+
}
|
|
927
|
+
}
|
|
928
|
+
|
|
690
929
|
function cleanAIPromptPrefix(text) {
|
|
691
930
|
if (!text) return "";
|
|
692
|
-
let cleaned = text
|
|
931
|
+
let cleaned = text;
|
|
932
|
+
// If text is a full valid JSON object or array, preserve structure
|
|
933
|
+
if (!isJsonPayload(cleaned)) {
|
|
934
|
+
// 1. Strip raw CSS / style blocks leaked from ChatGPT Canvas, web components or stylesheets (handles multi-line, unclosed and variable definitions)
|
|
935
|
+
cleaned = cleaned.replace(/^\s*(?:[.#][a-zA-Z0-9_-]+|\[[a-zA-Z0-9_#.:\-*>=,'"\s]+\]|:is\([^)]+\)|[a-zA-Z0-9_-]+)?\s*\{[^}]*?(?:\}\s*|\n\n+|$)/gi, "");
|
|
936
|
+
cleaned = cleaned.replace(/^[;{} \t\r\n]+/, "");
|
|
937
|
+
cleaned = cleaned.replace(/(?:^|\n)[a-zA-Z0-9_#.:\-*>[\]=\s,'"]+\{[^}]*(--[a-zA-Z0-9_-]+:|color-mix\(|var\()[^}]*\}/g, "");
|
|
938
|
+
}
|
|
939
|
+
// 2. Strip AI author prefixes and platform artifacts
|
|
940
|
+
cleaned = cleaned.replace(/^\s*(?:Claude responded|Claude|ChatGPT|Gemini|Grok|DeepSeek|Kimi|Copilot|Assistant|User)\s*(?::|\bsaid\b|\bresponded\b|(?=\s))\s*/i, "");
|
|
693
941
|
cleaned = cleaned.replace(/^(?:Here (?:is|are) (?:the )?(?:redacted|scrubbed|sanitized|processed|clean|updated|modified) (?:text|output|version|data).*?[:\n]+|\*\*Scrubbed Text\*\*[:\n]+|### Scrubbed Text[:\n]+)/i, '');
|
|
694
942
|
cleaned = cleaned.replace(/^\s*Edit\s*\n+/i, "");
|
|
695
943
|
cleaned = cleaned.replace(/\s*\bEdit\s+in\s+a\s+page\b\s*$/i, "");
|
|
944
|
+
// 3. Strip stray leading colons, semicolons, or separators left by stripped icons/artifact headers
|
|
945
|
+
cleaned = cleaned.replace(/^[:;|\-\—\–]+(?=\n|$)/, "");
|
|
946
|
+
cleaned = cleaned.replace(/^[:;]+\s*/, "");
|
|
696
947
|
return cleaned.trim();
|
|
697
948
|
}
|
|
698
949
|
|
|
950
|
+
function buildFastTokenLookup(sessionMap) {
|
|
951
|
+
const lookup = new Map();
|
|
952
|
+
const customRegexParts = [];
|
|
953
|
+
const keys = Object.keys(sessionMap || {});
|
|
954
|
+
|
|
955
|
+
for (let i = 0; i < keys.length; i++) {
|
|
956
|
+
const k = keys[i];
|
|
957
|
+
const v = sessionMap[k];
|
|
958
|
+
lookup.set(k, v);
|
|
959
|
+
lookup.set(k.toUpperCase(), v);
|
|
960
|
+
|
|
961
|
+
const inner = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
|
|
962
|
+
const match = inner.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
|
|
963
|
+
if (match) {
|
|
964
|
+
const label = match[1];
|
|
965
|
+
const baseIndex = parseInt(match[2], 10);
|
|
966
|
+
const aliases = getLabelAliases(label);
|
|
967
|
+
for (let a = 0; a < aliases.length; a++) {
|
|
968
|
+
const u = aliases[a].toUpperCase();
|
|
969
|
+
lookup.set(u + '_' + baseIndex, v);
|
|
970
|
+
lookup.set(u + '-' + baseIndex, v);
|
|
971
|
+
lookup.set(u + ' ' + baseIndex, v);
|
|
972
|
+
lookup.set(u + baseIndex, v);
|
|
973
|
+
lookup.set('[' + u + '_' + baseIndex + ']', v);
|
|
974
|
+
lookup.set('<' + u + '_' + baseIndex + '>', v);
|
|
975
|
+
lookup.set('{{' + u + '_' + baseIndex + '}}', v);
|
|
976
|
+
lookup.set('__' + u + '_' + baseIndex + '__', v);
|
|
977
|
+
lookup.set('[' + u + ' ' + baseIndex + ']', v);
|
|
978
|
+
lookup.set('[' + u + '-' + baseIndex + ']', v);
|
|
979
|
+
}
|
|
980
|
+
} else {
|
|
981
|
+
customRegexParts.push(k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
|
982
|
+
}
|
|
983
|
+
}
|
|
984
|
+
|
|
985
|
+
let regexStr = '(?:\\[\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*\\]|<\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*>|\\{\\{\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*\\}\\}|__\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*__|(?<=^|[^a-zA-Z0-9_А-Яа-яЁё])[A-Za-z_А-Яа-яЁё]+[-_\\s]*\\d+)';
|
|
986
|
+
if (customRegexParts.length > 0) {
|
|
987
|
+
regexStr = '(?:' + regexStr + '|' + customRegexParts.join('|') + ')';
|
|
988
|
+
}
|
|
989
|
+
const tokenRegex = new RegExp(regexStr + '(?:\'s|’s|s|[а-яёА-ЯЁ]{1,3})?', 'gi');
|
|
990
|
+
|
|
991
|
+
return { lookup, tokenRegex };
|
|
992
|
+
}
|
|
993
|
+
|
|
994
|
+
function resolveTokenValue(rawMatch, targetTokenKey, sessionMap) {
|
|
995
|
+
if (!sessionMap) return undefined;
|
|
996
|
+
if (sessionMap[targetTokenKey] !== undefined) return sessionMap[targetTokenKey];
|
|
997
|
+
if (sessionMap[rawMatch] !== undefined) return sessionMap[rawMatch];
|
|
998
|
+
const cleanRaw = rawMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
999
|
+
for (const k of Object.keys(sessionMap)) {
|
|
1000
|
+
const cleanK = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
1001
|
+
if (cleanK.toLowerCase() === cleanRaw.toLowerCase()) {
|
|
1002
|
+
return sessionMap[k];
|
|
1003
|
+
}
|
|
1004
|
+
}
|
|
1005
|
+
return undefined;
|
|
1006
|
+
}
|
|
1007
|
+
|
|
699
1008
|
function unscrubText(text, sessionMap) {
|
|
700
1009
|
let restoredCount = 0;
|
|
701
1010
|
let result = text;
|
|
702
|
-
const tokens = Object.keys(sessionMap);
|
|
1011
|
+
const tokens = Object.keys(sessionMap || {});
|
|
703
1012
|
if (tokens.length === 0) return { text: result, count: 0 };
|
|
704
1013
|
|
|
705
1014
|
result = cleanAIPromptPrefix(result);
|
|
706
1015
|
|
|
1016
|
+
if (tokens.length > 50) {
|
|
1017
|
+
const { lookup, tokenRegex } = buildFastTokenLookup(sessionMap);
|
|
1018
|
+
result = result.replace(tokenRegex, (match) => {
|
|
1019
|
+
if (lookup.has(match)) {
|
|
1020
|
+
restoredCount++;
|
|
1021
|
+
return lookup.get(match);
|
|
1022
|
+
}
|
|
1023
|
+
const upper = match.toUpperCase();
|
|
1024
|
+
if (lookup.has(upper)) {
|
|
1025
|
+
restoredCount++;
|
|
1026
|
+
return lookup.get(upper);
|
|
1027
|
+
}
|
|
1028
|
+
const possMatch = match.match(/^([\s\S]+?)('s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1029
|
+
if (possMatch) {
|
|
1030
|
+
const base = possMatch[1];
|
|
1031
|
+
const suffix = possMatch[2];
|
|
1032
|
+
if (lookup.has(base)) {
|
|
1033
|
+
restoredCount++;
|
|
1034
|
+
return lookup.get(base) + suffix;
|
|
1035
|
+
}
|
|
1036
|
+
if (lookup.has(base.toUpperCase())) {
|
|
1037
|
+
restoredCount++;
|
|
1038
|
+
return lookup.get(base.toUpperCase()) + suffix;
|
|
1039
|
+
}
|
|
1040
|
+
}
|
|
1041
|
+
return match;
|
|
1042
|
+
});
|
|
1043
|
+
return { text: result, count: restoredCount };
|
|
1044
|
+
}
|
|
1045
|
+
|
|
707
1046
|
const { compositeRegex, looseRules } = buildRestorationRegexAndRules(sessionMap);
|
|
708
1047
|
|
|
709
1048
|
if (compositeRegex) {
|
|
710
1049
|
result = result.replace(compositeRegex, (match) => {
|
|
711
|
-
const
|
|
712
|
-
const
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
1050
|
+
const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1051
|
+
const suffix = suffixMatch ? suffixMatch[0] : '';
|
|
1052
|
+
const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
|
|
1053
|
+
const cleanMatch = baseMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
1054
|
+
const token = `[${cleanMatch}]`;
|
|
1055
|
+
|
|
1056
|
+
const val = resolveTokenValue(baseMatch, token, sessionMap) ?? resolveTokenValue(baseMatch, cleanMatch, sessionMap);
|
|
1057
|
+
if (val !== undefined) {
|
|
718
1058
|
restoredCount++;
|
|
719
|
-
return
|
|
1059
|
+
return val + suffix;
|
|
720
1060
|
}
|
|
721
1061
|
return match;
|
|
722
1062
|
});
|
|
@@ -725,8 +1065,16 @@ const PROFILE_JARGON = {
|
|
|
725
1065
|
looseRules.forEach(rule => {
|
|
726
1066
|
const rx = new RegExp(rule.pattern, 'gi');
|
|
727
1067
|
result = result.replace(rx, (match) => {
|
|
728
|
-
|
|
729
|
-
|
|
1068
|
+
const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1069
|
+
const suffix = suffixMatch ? suffixMatch[0] : '';
|
|
1070
|
+
const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
|
|
1071
|
+
|
|
1072
|
+
const val = sessionMap[rule.token] ?? resolveTokenValue(baseMatch, rule.token, sessionMap);
|
|
1073
|
+
if (val !== undefined) {
|
|
1074
|
+
restoredCount++;
|
|
1075
|
+
return val + suffix;
|
|
1076
|
+
}
|
|
1077
|
+
return match;
|
|
730
1078
|
});
|
|
731
1079
|
});
|
|
732
1080
|
|
|
@@ -750,21 +1098,50 @@ const PROFILE_JARGON = {
|
|
|
750
1098
|
const tokens = Object.keys(sessionMap || {});
|
|
751
1099
|
if (tokens.length === 0) return { text: result, count: 0 };
|
|
752
1100
|
|
|
1101
|
+
if (tokens.length > 50) {
|
|
1102
|
+
const { lookup, tokenRegex } = buildFastTokenLookup(sessionMap);
|
|
1103
|
+
result = result.replace(tokenRegex, (match) => {
|
|
1104
|
+
let rawVal = null;
|
|
1105
|
+
let suffix = '';
|
|
1106
|
+
if (lookup.has(match)) {
|
|
1107
|
+
rawVal = lookup.get(match);
|
|
1108
|
+
} else if (lookup.has(match.toUpperCase())) {
|
|
1109
|
+
rawVal = lookup.get(match.toUpperCase());
|
|
1110
|
+
} else {
|
|
1111
|
+
const possMatch = match.match(/^([\s\S]+?)('s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1112
|
+
if (possMatch) {
|
|
1113
|
+
const base = possMatch[1];
|
|
1114
|
+
suffix = possMatch[2];
|
|
1115
|
+
if (lookup.has(base)) rawVal = lookup.get(base);
|
|
1116
|
+
else if (lookup.has(base.toUpperCase())) rawVal = lookup.get(base.toUpperCase());
|
|
1117
|
+
}
|
|
1118
|
+
}
|
|
1119
|
+
if (rawVal !== null) {
|
|
1120
|
+
restoredCount++;
|
|
1121
|
+
const escapedVal = escapeHTML(rawVal);
|
|
1122
|
+
const cleanMatch = match.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
1123
|
+
return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(cleanMatch)}">${escapedVal}</span>${escapeHTML(suffix)}`;
|
|
1124
|
+
}
|
|
1125
|
+
return match;
|
|
1126
|
+
});
|
|
1127
|
+
return { text: result, count: restoredCount };
|
|
1128
|
+
}
|
|
1129
|
+
|
|
753
1130
|
const { compositeRegex, looseRules } = buildRestorationRegexAndRules(sessionMap);
|
|
754
1131
|
|
|
755
1132
|
if (compositeRegex) {
|
|
756
1133
|
result = result.replace(compositeRegex, (match) => {
|
|
757
|
-
const
|
|
758
|
-
const
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
if (
|
|
1134
|
+
const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1135
|
+
const suffix = suffixMatch ? suffixMatch[0] : '';
|
|
1136
|
+
const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
|
|
1137
|
+
const cleanMatch = baseMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
|
|
1138
|
+
const token = `[${cleanMatch}]`;
|
|
1139
|
+
|
|
1140
|
+
const val = resolveTokenValue(baseMatch, token, sessionMap) ?? resolveTokenValue(baseMatch, cleanMatch, sessionMap);
|
|
1141
|
+
if (val !== undefined) {
|
|
765
1142
|
restoredCount++;
|
|
766
|
-
const escapedVal = escapeHTML(
|
|
767
|
-
return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(
|
|
1143
|
+
const escapedVal = escapeHTML(val);
|
|
1144
|
+
return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(token)}">${escapedVal}</span>${escapeHTML(suffix)}`;
|
|
768
1145
|
}
|
|
769
1146
|
return match;
|
|
770
1147
|
});
|
|
@@ -778,9 +1155,17 @@ const PROFILE_JARGON = {
|
|
|
778
1155
|
const lastClose = before.lastIndexOf('>');
|
|
779
1156
|
if (lastOpen > lastClose) return match;
|
|
780
1157
|
|
|
781
|
-
|
|
782
|
-
const
|
|
783
|
-
|
|
1158
|
+
const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
|
|
1159
|
+
const suffix = suffixMatch ? suffixMatch[0] : '';
|
|
1160
|
+
const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
|
|
1161
|
+
|
|
1162
|
+
const val = sessionMap[rule.token] ?? resolveTokenValue(baseMatch, rule.token, sessionMap);
|
|
1163
|
+
if (val !== undefined) {
|
|
1164
|
+
restoredCount++;
|
|
1165
|
+
const escapedVal = escapeHTML(val);
|
|
1166
|
+
return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(rule.token)} (Fuzzy Match)">${escapedVal}</span>${escapeHTML(suffix)}`;
|
|
1167
|
+
}
|
|
1168
|
+
return match;
|
|
784
1169
|
});
|
|
785
1170
|
});
|
|
786
1171
|
|
|
@@ -798,6 +1183,7 @@ const PROFILE_JARGON = {
|
|
|
798
1183
|
LABEL_ALIASES,
|
|
799
1184
|
getLabelAliases,
|
|
800
1185
|
PROFILE_ALIAS_MAP,
|
|
1186
|
+
formatToken,
|
|
801
1187
|
buildRestorationRegexAndRules,
|
|
802
1188
|
unscrubText,
|
|
803
1189
|
unscrubTextAsHTML,
|