@privacyscrubber/mcp-server 1.7.6 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/ps-pii-engine.cjs CHANGED
@@ -19,7 +19,9 @@ let DEVOPS_SECRETS = [
19
19
  { name: 'JSON Web Token (JWT)', type: 'SECRET', regex: /\beyJ[a-zA-Z0-9_-]+\.[a-zA-Z0-9_-]+\.[a-zA-Z0-9_-]+\b/g },
20
20
  { name: 'API Token/Key (GitHub/Slack/NPM)', type: 'SECRET', regex: /\b(?:ghp|gho|ghu|ghs|ghr|glpat|npm|xox[baprs])[-_][A-Za-z0-9_]{10,}\b/g },
21
21
  { name: 'Stripe API Key', type: 'SECRET', regex: /\b(?:[rs]k)_(?:test|live)_[a-zA-Z0-9]{24,}\b/g },
22
- { name: 'Generic Secret/Key', type: 'SECRET', regex: /\b(sk|pk|secret|key|token|auth)(?:[-_][a-zA-Z0-9_-]{5,}|(?=[a-zA-Z0-9_-]{5,}\b)(?=[a-zA-Z_-]*[0-9])[a-zA-Z0-9_-]{5,})\b/gi },
22
+ { name: 'OpenAI Project API Key', type: 'SECRET', regex: /\b(?:sk|pk)-(?:proj-)?[a-zA-Z0-9_-]{16,}\b/gi },
23
+ { name: 'Database Connection URI', type: 'SECRET', regex: /\b(?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp|mssql):\/\/[^\s"']+/gi },
24
+ { name: 'Generic Secret/Key', type: 'SECRET', regex: /\b(sk|pk|secret|key|token|auth)(?:[-_][a-zA-Z0-9_-]{3,}|(?=[a-zA-Z0-9_-]{5,}\b)(?=[a-zA-Z_-]*[0-9])[a-zA-Z0-9_-]{5,})\b/gi },
23
25
  { name: 'Hash / Hex Key (32-64 chars)', type: 'SECRET', regex: /\b[a-fA-F0-9]{32,64}\b/g },
24
26
  { name: 'CVE Identifier', type: 'SECRET', regex: /\bCVE-\d{4}-\d{4,}\b/gi },
25
27
  { name: 'Cryptographic Hash', type: 'SECRET', regex: /\b(MD5|SHA1|SHA256)[:\s][a-f0-9]{32,64}\b/gi },
@@ -31,25 +33,29 @@ let DEVOPS_SECRETS = [
31
33
  let REGEX_RULES = [
32
34
  ...DEVOPS_SECRETS,
33
35
  // Emails
34
- { type: 'EMAIL', regex: /[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}/g },
36
+ { type: 'EMAIL', regex: /\b[a-zA-Z0-9._%+-]{1,64}@[a-zA-Z0-9.-]{1,255}\.[a-zA-Z]{2,}\b/g },
35
37
 
36
- // Financial Data
37
- { type: 'FINANCIAL', regex: /(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)[0-9,.]+\b/g },
38
- { type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9]([A-Z0-9]{3})?\b/g },
39
- { type: 'FINANCIAL', regex: /\b\d{9}\b/g },
38
+ // Financial Data (PCI-DSS, Bank Accounts, Direct Deposits, Cards, IBAN, SWIFT, Routing Numbers)
39
+ { type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9](?:[A-Z0-9]{3})?\b/g },
40
40
  { type: 'FINANCIAL', regex: /\b[A-Z]{2}[0-9]{2}[a-zA-Z0-9]{4}[0-9]{7}[a-zA-Z0-9]{0,16}\b/g },
41
41
  { type: 'FINANCIAL', regex: /\bPORTFOLIO[-_][A-Z0-9]{5,}\b/gi },
42
42
  { type: 'FINANCIAL', regex: /\b(?:\d[ -]?){13,19}\b/g },
43
43
  { type: 'FINANCIAL', regex: /\b(?:1|3|bc1)[a-zA-HJ-NP-Z0-9]{25,39}\b/g },
44
44
  { type: 'FINANCIAL', regex: /\b0x[a-fA-F0-9]{40}\b/g },
45
+ // Masked / Direct Deposit Bank Account Numbers & Routing Numbers
46
+ { type: 'FINANCIAL', regex: /\b(?:Account|Acct|Checking|Savings|Direct\s+Deposit)\s*(?:#|ID|No\.?|Number)?[:\s#]*(?:[\*xX•.-]{3,}\d{2,6}|\d{4}[-\s]?\d{4}[-\s]?\d{2,6})\b/gi },
47
+ { type: 'FINANCIAL', regex: /\b(?:ABA|Routing|RTN)\s*(?:#|ID|No\.?|Number)?[:\s#]*\d{9}\b/gi },
45
48
 
46
49
  // Legal & Court
47
- { type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9]{4,}\b/gi },
48
- { type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9]{4,}\b/gi },
49
- { type: 'LEGAL', regex: /\b[A-Z]{2,4}[- ]?\d{2}[- ]?\d{4,}\b/g },
50
+ { type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9_-]{4,}\b/gi },
51
+ { type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9_-]{4,}\b/gi },
50
52
  { type: 'PRIVILEGE', regex: /ATTORNEY[- ]CLIENT[- ]PRIVILEGE/gi },
51
53
 
52
- // Professional IDs
54
+ // Professional IDs & Organizations
55
+ { type: 'NAME', isContextName: true, regex: /\b(?:[A-Z][A-Za-z0-9&.,'-]*[ \t\xA0]+){1,5}(?:Inc\.?|LLC|Corp\.?|Corporation|Ltd\.?|Limited|Co\.?|Company|Group|Holdings|Solutions|Services|Technologies|Logistics|Industries|Capital|Bank|Partners|LLP|PLLC)(?:\s+(?:LLC|Inc\.?|Corp\.?|Ltd\.?|USA|Group))?\b/g },
56
+ { type: 'ID', regex: /\b(?:Employee|Emp|EE|Worker|Staff|File|Badge|Member|Advisor|Producer|Agent|Borrower)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Z0-9-]{3,15}\b/gi },
57
+ { type: 'ID', regex: /\b(?:Pay\s+Group|Cost\s+Center|Dept|Department)[:\s#]*[A-Za-z0-9_-]{2,30}\b/gi },
58
+ { type: 'ID', regex: /(?:\b(?:Box\s+d\b|d\.\s*(?:Control|#)?|d\s+Control)\s*(?:number|no\.?|#|num)?[:\s#]*|\bControl\s*(?:number|no\.?|#|num)[:\s#]*|\bControl[:#]\s*)([A-Za-z0-9-]{3,30})/gi },
53
59
  { type: 'ID', regex: /\bEEID[ -]?\d{4,}\b/gi },
54
60
  { type: 'ID', regex: /\bRESUME[-_]?[A-Z0-9]{4,}\b/gi },
55
61
  { type: 'ID', regex: /\bLEAD[-_][A-Z0-9]{5,}\b/gi },
@@ -69,34 +75,45 @@ let REGEX_RULES = [
69
75
  { type: 'ID', regex: /\bCOURSE[-_][A-Z0-9]{4,}\b/gi },
70
76
  { type: 'ID', regex: /\bINSTANCE[-_]ID[-_][a-z0-9-]{10,}\b/gi },
71
77
  { type: 'ID', regex: /\bENV[-_][A-Z0-9]{3,}\b/gi },
72
- { type: 'PRIVACY', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
78
+ { type: 'ID', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
73
79
 
74
- // Addresses & Locations
75
- { type: 'ADDRESS', regex: /\b\d{1,6}\s+(?:[A-Z][a-zA-Z0-9.-]*\s+){1,3}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square)\b/gi },
76
- { type: 'LOCATION', regex: /\b[A-Z][a-zA-Z\s.-]{2,25},\s*[A-Z]{2}\b/g },
80
+ // Insurance & Health Plan IDs
81
+ { type: 'ID', regex: /\b(?:BCBS|AETNA|CIGNA|UHC|HUMANA|MEDICARE|MEDICAID)[-_A-Za-z0-9]+\b/gi },
82
+ { type: 'ID', regex: /\b(?:Insurance|Policy|Member|Subscriber|Group|Plan|Health|Rx)[-_: ]*ID[:\s#]*([A-Za-z0-9-]+)/gi },
77
83
 
84
+ // Addresses & Locations
85
+ { type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:\bBox\s+f\b|\bf\.\s*|\bf\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?(?:address[,\s]+and\s+ZIP\s+code|address)?|(?:Employee(?:'s)?\s+address[,\s]+and\s+ZIP\s+code))[\s:#]*([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|Box|\d+\b|1\b|2\b|Wages|Federal|Social|Medicare)/gi },
86
+ { type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:Borrower(?:'s)?|Co-Borrower(?:'s)?|Employee(?:'s)?|Employer(?:'s)?|Home|Mailing|Property|Physical)\s+address)[\s:#]+([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|City|State|ZIP|SSN|EIN|Phone|Box|\d+\b)/gi },
87
+ { type: 'ADDRESS', regex: /\b\d{1,6}[ \t\xA0]+(?:[A-Za-z0-9.-]+[ \t\xA0]+){1,4}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Hwy|Highway|Cir|Circle|Trl|Trail|Loop|Row|Pike|Box|PO Box|P\.O\.[ \t\xA0]*Box)\b(?:[ \t\xA0]*,?[ \t\xA0]*(?:Apt|Apartment|Suite|Ste|Unit|#|Fl|Floor|Bldg|Building)\.?[ \t\xA0]*[A-Za-z0-9-]+)?/gi },
88
+ { type: 'ADDRESS', regex: /\b(?:P\.?O\.?[ \t\xA0]*Box|PO[ \t\xA0]*Box)[ \t\xA0]+\d{1,6}\b/gi },
89
+ { type: 'ADDRESS', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
90
+ { type: 'ADDRESS', regex: /\b(?:ZIP|Postal|Code)?\s*(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
91
+ { type: 'ADDRESS', regex: /\b\d{5}-\d{4}\b/g },
92
+ { type: 'LOCATION', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\b/g },
78
93
 
79
94
  // PHI & Medical
95
+ { type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record No|Patient No)[\s:#]+([A-Za-z0-9-]+)/gi },
80
96
  { type: 'PHI', regex: /\bMRN[ -]?\d{6,}\b/gi },
81
97
  { type: 'PHI', regex: /\b[A-TV-Z]\d{2}[. ]?\d[A-Z0-9]?\b/g },
82
98
  { type: 'PHI', regex: /\b[A-Z]{2,3}\d{6,8}\b/g },
83
99
  { type: 'PHI', regex: /\bNHS[ -]?\d{3}[ -]?\d{3}[ -]?\d{4}\b/gi },
84
100
 
85
-
86
101
  // Copyrights
87
102
  { type: 'COPYRIGHT', regex: /\bPROJECT[-_][A-Z0-9]{5,}\b/gi },
88
103
  { type: 'COPYRIGHT', regex: /\b(DRAFT|ASSET|SCRIPT)[-_][0-9]{4,}\b/gi },
89
104
 
90
- // General Privacy & Dates
91
- { type: 'PRIVACY', regex: /\b(GDPR|HIPAA|CCPA|SOC2)[-_]AUDIT[-_]\d{4}\b/gi },
92
- { type: 'PRIVACY', regex: /\bPOLICY[-_][A-Z0-9]{5,}\b/gi },
93
- { type: 'PRIVACY', regex: /\bGRADE[S]?\s*:\s*[A-DF][+-]?\b/gi },
94
- { type: 'PRIVACY', regex: /\b(DOB|BIRTHDAY)[:\s]*[0-9./-]{6,10}\b/gi },
95
- { type: 'PRIVACY', regex: /\b(PASSWORD|PWD|SECRET)[:\s]*[\S]{4,}\b/gi },
105
+ // General Privacy, Dates & Secrets
106
+ { type: 'ID', regex: /\b(?:GDPR|HIPAA|CCPA|SOC2)[-_]AUDIT[-_]\d{4}\b/gi },
107
+ { type: 'ID', regex: /\bPOLICY[-_][A-Z0-9]{5,}\b/gi },
108
+ { type: 'ID', regex: /\bGRADE[S]?\s*:\s*[A-DF][+-]?\b/gi },
109
+ { type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
110
+ { type: 'SECRET', regex: /\b(?:PASSWORD|PWD|SECRET)\s*[:=]\s*["']?[\S]{4,}["']?/gi },
96
111
 
97
112
  // Standard IDs (SSN, EIN, Passport, VAT)
98
113
  { type: 'ID', regex: /\b\d{3}-\d{2}-\d{4}\b/g },
99
- { type: 'ID', regex: /(?:\/|[A-Za-z]:\\)[\w\-. ]+(?:[\/\\][\w\-. ]+)+/g },
114
+ { type: 'ID', regex: /\b(?:XXX|xxx|\*\*\*)[ -]?(?:XX|xx|\*\*)[ -]?\d{4}\b/g },
115
+ { type: 'ID', regex: /\b\d{2}-\d{7}\b/g },
116
+ { type: 'ID', regex: /(?:(?:[A-Za-z]:\\|\/(?:usr|var|etc|home|root|Users|private|tmp|opt|bin|sbin|dev|Applications|Library)\/)[a-zA-Z0-9_.-]+(?:[\/\\][a-zA-Z0-9_.-]+)*|\/(?:[a-zA-Z0-9_.-]+\/)+[a-zA-Z0-9_.-]+\.(?:txt|pdf|docx|xlsx|csv|js|ts|json|env|log|key|pem|crt|conf|yaml|yml|xml|html|sql|py|go|rs|c|cpp|h|sh|bin|zip|tar|gz|png|jpg|jpeg|svg|webp|wasm)\b)/g },
100
117
  { type: 'ID', regex: /\b[A-CEGHJ-PR-TW-Z]{1}[A-CEGHJ-NPR-TW-Z]{1}[0-9]{6}[A-DFM]{1}\b/gi },
101
118
  { type: 'ID', regex: /\b[A-Z]{2}[0-9]{6,12}\b/gi },
102
119
  { type: 'ID', regex: /[A-Z0-9<]{30,44}/g },
@@ -105,14 +122,13 @@ let REGEX_RULES = [
105
122
  { type: 'IP', regex: /\b(?:\d{1,3}\.){3}\d{1,3}\b/g },
106
123
  { type: 'IP', regex: /\b(?:[a-fA-F0-9]{1,4}:){7}[a-fA-F0-9]{1,4}\b/g },
107
124
  { type: 'ID', regex: /\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b/g },
108
- { type: 'ID', regex: /(?:\B\/|\b[a-zA-Z]:\\)(?:[\w.-]+[\/\\]?)+\b/g },
125
+ { type: 'ID', regex: /(?:\B\/|\b[a-zA-Z]:\\)(?:[\w.-]+[\/\\])*[\w.-]+\b/g },
109
126
  { type: 'ID', regex: /\b\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}(?::\d{2})?)?\b/g },
110
127
  { type: 'ID', regex: /\b\d{4}-\d{2}-\d{2}\b/g },
111
128
  { type: 'ID', regex: /\b\d{2}\/\d{2}\/\d{4}\b/g },
112
129
 
113
- // Phone Numbers — must come after IDs to avoid ID rules stealing phone matches
114
- { type: 'PHONE', regex: /\b(?:\+?\d{1,3}[-.\s]?)?\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}\b/g },
115
- { type: 'PHONE', regex: /(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}/g },
130
+ // Phone Numbers
131
+ { type: 'PHONE', regex: /(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}\b/g },
116
132
  { type: 'PHONE', regex: /\+?[1-9]\d{1,3}[\s.-]\(?\d{1,4}\)?[\s.-]\d{2,4}[\s.-]\d{4}/g },
117
133
  { type: 'PHONE', regex: /(?:\+44\s?7\d{3}|\(?07\d{3}\)?)\s?\d{3}\s?\d{3}\b/g },
118
134
  { type: 'PHONE', regex: /\b(?:\d{3}[-.\s]\d{4}|\(\d{3}\)\s??\d{3}[-.\s]??\d{4}|\d{3}[-.\s]??\d{3}[-.\s]??\d{4})\b/g },
@@ -126,50 +142,63 @@ let REGEX_RULES = [
126
142
  { type: 'ID', regex: /\bDRIVER[S]?\s+LICENSE[ -]?\d{6,15}\b/gi },
127
143
 
128
144
  // Generalized Name Detection (First [Middle] Last) — max 1 middle word to avoid grabbing job titles
129
- // Middle word must be: a particle (van/de/etc.), a single initial (A.), or a capitalized word of ≥2 lowercase letters
130
- { type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]{1,}(?:\p{Lu}[\p{Ll}'-]*)?\p{L}|\p{Lu}\.?|van|von|de|di|da|la|le|del|du|der|van[ \t\xA0]+de)){0,1}[ \t\xA0]+\p{Lu}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
131
- // All-Caps Names (2–3 words, each ≥3 chars) — prevents acronym chains like "GPA ROI AUM" (Unicode-safe)
132
- { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{3,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+\p{Lu}{3,}(?:[\p{Lu}'-]*\p{Lu})?){1,2}(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
133
- // ALL-CAPS first + optional middle initial(s) with optional spaces + ALL-CAPS last name
134
- // E.g. "KKKKK I. MMMMMM", "KKKKK I.MMMMMM", "KKKKKI.MMMMMM" (PDF extraction spacing anomalies)
135
- { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*|[ \t\xA0]+)\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
145
+ // Middle word must be: a particle (van/de/etc.), a single initial (A. or A), or a capitalized word of ≥2 lowercase letters
146
+ { type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*|\p{Lu}\.?|van|von|de|di|da|la|le|del|du|der|van[ \t\xA0]+de|van[ \t\xA0]+der)){0,1}[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
147
+ // Payroll Format Names (e.g. "BARKER, KELLY", "BARKER, KELLY M", "DOE, JOHN M.")
148
+ { type: 'NAME', regex: /\b[A-Z]{2,25},\s+[A-Z]{2,25}(?:\s+[A-Z]\.?|\s+[A-Z]{2,25})*\b/g },
149
+ // All-Caps Names (2–3 words, supporting single-letter middle initial e.g. "KELLY M BARKER", "JOHN M BARKER")
150
+ { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+(?:\p{Lu}\.?|[A-Z][a-z]+))?[ \t\xA0]+\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
151
+ // ALL-CAPS first + middle initial(s) with optional spaces + ALL-CAPS last name
152
+ { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
136
153
  // ALL-CAPS first name + optional middle initial(s) with optional spaces + Mixed-Case last name
137
- // E.g. "JOSEPH I. Casalvieri", "JOSEPH I.Casalvieri", "JOSEPHI.Casalvieri" (PDF extraction spacing anomalies)
138
- { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*|[ \t\xA0]+)\p{Lu}\p{Ll}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
139
- // Single-Word Names with Honorifics (with or without period) (Unicode-safe)
140
- { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])(?:Mr|Mrs|Ms|Dr|Prof|Hon|Mr\.|Mrs\.|Ms\.|Dr\.|Prof\.|Hon\.)[ \t\xA0]+\p{Lu}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
154
+ { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*|[ \t\xA0]+)(?:\p{Lu}\p{Ll}[\p{Ll}'-]*|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
155
+ // Names with Honorifics (with or without period, supporting single or multi-word full names) (Unicode-safe)
156
+ { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])(?:Mr|Mrs|Ms|Dr|Prof|Hon|Mr\.|Mrs\.|Ms\.|Dr\.|Prof\.|Hon\.)[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*))?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
141
157
 
142
- // Contextual Names (Emergency Contacts, Spouses, Children, Tenants, etc.) - Uses capture group m[1]
143
- { type: 'NAME', isContextName: true, regex: /(?:Emergency Contact|Contact(?: Name)?|Spouse|Child|Parent|Guardian|Relationship|Kin|Tenant|Landlord|Buyer|Seller|Plaintiff|Defendant|Testator)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g }
158
+ // W-2 Box c Employer Block (Name, Address, and Zip Code)
159
+ { type: 'NAME', isContextName: true, regex: /(?:(?:\bBox\s+c\b|\bc\.\s*|\bc\s+(?=Employer))\s*(?:Employer(?:'s)?\s*)?(?:name[,\s]+address[,\s]+and\s+ZIP\s+code|name)?|(?:Employer(?:'s)?\s+name[,\s]+address[,\s]+and\s+ZIP\s+code))[\s:#]*([A-Za-z0-9&., \t\xA0'-]{2,45}?)(?=\r?\n|$|\s{3,}|\t|EIN|FEIN|Box|\d+\b|Wages|Federal|Social|Medicare)/gi },
160
+ // W-2 Box e Employee Name & Initial
161
+ { type: 'NAME', isContextName: true, regex: /(?:(?:\bBox\s+e\b|\be\.\s*|\be\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?(?:first\s+name(?:\s+(?:and|&)\s+initial)?|name)?[:\s#]*|(?:Employee(?:'s)?\s+first\s+name(?:\s+(?:and|&)\s+initial)?))[\s:#]*([A-Za-z0-9.\s'-]{2,35}?)(?=\r?\n|$|\s{3,}|\t|Last|Surname|Suff|Box|\d+\b|1\b)/gi },
162
+ // Contextual First Names (Employee's first name, First name, Given name)
163
+ { type: 'NAME', isContextName: true, regex: /(?:(?:Employee(?:'s)?|Borrower(?:'s)?|Co-Borrower(?:'s)?|Applicant(?:'s)?|Candidate(?:'s)?|Worker(?:'s)?|Taxpayer(?:'s)?|Spouse(?:'s)?|Person(?:'s)?)\s+)?(?:First\s+name(?:\s+(?:and|&)\s+initial)?|Given\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9.\s'-]{2,30}?)(?=\r?\n|$|\s{3,}|\t|Last|Surname|Family|Suff|Box|Address|SSN|EIN)/gi },
164
+ // Contextual Last Names (Last name, Surname, Family name)
165
+ { type: 'NAME', isContextName: true, regex: /(?:(?:Employee(?:'s)?|Borrower(?:'s)?|Co-Borrower(?:'s)?|Applicant(?:'s)?|Candidate(?:'s)?|Worker(?:'s)?|Taxpayer(?:'s)?|Spouse(?:'s)?|Person(?:'s)?)\s+)?(?:Last\s+name|Surname|Family\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z'-]{2,30})/gi },
166
+ // Contextual General Names (Employee, Borrower, Co-Borrower, Taxpayer, Spouse, Applicant, Candidate, Worker, Employer, Company, Insured, Patient, Client, etc.)
167
+ { type: 'NAME', isContextName: true, regex: /(?:Employee(?:\s+Name)?|Employer(?:\s+Name)?|Borrower(?:\s+Name)?|Co-Borrower(?:\s+Name)?|Applicant(?:\s+Name)?|Candidate(?:\s+Name)?|Worker(?:\s+Name)?|Taxpayer(?:\s+Name)?|Spouse(?:\s+Name)?|Manager|Supervisor|Reporting To|Insured|Claimant|Patient|Client|Customer|Account Holder|Prepared By|Attention|Attn|Contact(?: Name)?|Child|Parent|Guardian|Relationship|Kin|Tenant|Landlord|Buyer|Seller|Plaintiff|Defendant|Testator)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9&.,\s'-]{2,40}?)(?=\r?\n|$|\s{3,}|\t|Employee|Employer|Address|Phone|SSN|EIN|FEIN|Date|Pay|Rate|Tax|W-2|OMB|Copy|Box|Status)/gi },
168
+ // Box e shorthand
169
+ { type: 'NAME', isContextName: true, regex: /\b(?:Box\s+e)\s*[:#-]\s*([A-Za-z0-9&.,\s'-]{2,40})/gi }
144
170
  ];
145
171
 
146
172
  let PROFILE_RULES = {
147
173
  general: [],
148
174
  legal: [
149
- { type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9]{4,}\b/gi },
150
- { type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9]{4,}\b/gi },
175
+ { type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9_-]{4,}\b/gi },
176
+ { type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9_-]{4,}\b/gi },
151
177
  { type: 'LEGAL', regex: /\b[A-Z]{2,4}[- ]?\d{2}[- ]?\d{4,}\b/g },
152
178
  { type: 'PRIVILEGE', regex: /ATTORNEY[- ]CLIENT[- ]PRIVILEGE/gi }
153
179
  ],
154
180
  hr: [
155
- { type: 'NAME', isContextName: true, regex: /(?:Candidate|Applicant|Employee|Reporting To|Manager)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g },
156
-
181
+ { type: 'NAME', isContextName: true, regex: /(?:Candidate|Applicant|Employee|Reporting To|Manager|Mentored by|Direct Report)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g },
157
182
  { type: 'ID', regex: /\bEEID[ -]?\d{4,}\b/gi },
158
183
  { type: 'ID', regex: /\bEMP[-_]\d{3,}\b/gi },
159
184
  { type: 'ID', regex: /\bRESUME[-_]?[A-Z0-9]{4,}\b/gi },
160
- { type: 'PRIVACY', regex: /\\b(DOB|BIRTHDAY)[:\\s]*[0-9./-]{6,10}\\b/gi },
161
- { type: 'ADDRESS', regex: /\\b\\d{1,6}\\s+(?:[A-Z][a-zA-Z0-9.-]*\\s+){1,3}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square)\\b/gi }
185
+ { type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
186
+ { type: 'DATE', regex: /\b(?:Graduated|Graduation|Class of)[:\s]+(?:(?:Spring|Summer|Fall|Winter)\s+)?(?:(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec|January|February|March|April|June|July|August|September|October|November|December)\s+)?\d{4}\b/gi },
187
+ { type: 'ID', regex: /(?:https?:\/\/)?(?:www\.)?linkedin\.com\/in\/[A-Za-z0-9_-]+/gi },
188
+ { type: 'ID', regex: /(?:https?:\/\/)?(?:www\.)?github\.com\/[A-Za-z0-9_-]+/gi },
189
+ { type: 'ADDRESS', regex: /\b\d{1,6}\s+(?:[A-Z0-9][a-zA-Z0-9-]*\s+){1,3}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Highway|Hwy|Circle|Cir|Trail|Trl|Dr(?!\.?\s+[A-Z][a-z]+))\b/g }
162
190
  ],
163
191
  finance: [
164
- { type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9]([A-Z0-9]{3})?\b/g },
192
+ { type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9](?:[A-Z0-9]{3})?\b/g },
165
193
  { type: 'FINANCIAL', regex: /\b[A-Z]{2}[0-9]{2}[a-zA-Z0-9]{4}[0-9]{7}[a-zA-Z0-9]{0,16}\b/g },
166
194
  { type: 'FINANCIAL', regex: /\bPORTFOLIO[-_][A-Z0-9]{5,}\b/gi }
167
195
  ],
168
196
  medical: [
169
- { type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record No|Patient No)[\s:#]*([A-Za-z0-9-]+)/gi },
170
- { type: 'PRIVACY', regex: /\b(DOB|Date of Birth)[:\s]*[0-9./-]{6,10}\b/gi },
171
-
197
+ { type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record No|Patient No)[\s:#]+([A-Za-z0-9-]+)/gi },
198
+ { type: 'DATE', regex: /\b(?:DOB|Date of Birth|BIRTHDAY)[\s:]+([0-9./-]{6,10})\b/gi },
172
199
  { type: 'PHI', regex: /\bMRN[ -]?\d{6,}\b/gi },
200
+ { type: 'ID', regex: /\b(?:Insurance|Policy|Member|Subscriber|Group|Plan|Health|Rx)[-_: ]*ID[:\s#]*([A-Za-z0-9-]+)/gi },
201
+ { type: 'ID', regex: /\b(?:BCBS|AETNA|CIGNA|UHC|HUMANA|MEDICARE|MEDICAID)[-_A-Za-z0-9]+\b/gi },
173
202
  { type: 'PHI', regex: /\b[A-TV-Z]\d{2}[. ]?\d[A-Z0-9]?\b/g },
174
203
  { type: 'PHI', regex: /\b[A-Z]{2,3}\d{6,8}\b/g },
175
204
  { type: 'PHI', regex: /\bNHS[ -]?\d{3}[ -]?\d{3}[ -]?\d{4}\b/gi }
@@ -186,13 +215,13 @@ let PROFILE_RULES = {
186
215
  bizops: [
187
216
  { type: 'ID', regex: /\b(?:DEAL|KPI|METRIC)[-_: ]?[A-Z0-9]{4,}\b/gi },
188
217
  { type: 'ID', regex: /\b(?:ENTITY|VENDOR|PARTNER)[-_: ]?[0-9]{4,10}\b/gi },
189
- { type: 'FINANCIAL', regex: /\b(?:REVENUE|EBITDA|PROFIT|MARGIN)[-_: ]?(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
218
+ { type: 'FINANCIAL', regex: /\b(?:REVENUE|EBITDA|PROFIT|MARGIN)[\s:_-]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
190
219
  { type: 'SECRET', regex: /\b(?:NDA|M&A|MERGER)[-_: ]?[A-Z0-9]{4,}\b/gi }
191
220
  ],
192
221
  sales: [
193
222
  { type: 'ID', regex: /\bOPPORTUNITY[-_: ]?[A-Z0-9]{5,}\b/gi },
194
223
  { type: 'ID', regex: /\b(?:DOCUSIGN|CONTRACT)[-_: ]?[0-9A-F]{8,32}\b/gi },
195
- { type: 'FINANCIAL', regex: /\b(?:ARR|MRR|QUOTA)[\s:]?(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
224
+ { type: 'FINANCIAL', regex: /\b(?:ARR|MRR|QUOTA)[\s:]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
196
225
  { type: 'ID', regex: /\b(?:SFDC|HUBSPOT)[-_: ]?[0-9A-Z]{15,18}\b/gi }
197
226
  ],
198
227
  support: [
@@ -205,8 +234,8 @@ let PROFILE_RULES = {
205
234
  { type: 'ID', regex: /\b(?:MLS|LIS)[- ]?\d{6,10}\b/gi },
206
235
  { type: 'ID', regex: /\bPARCEL[- ]?\d{5,15}\b/gi },
207
236
  { type: 'ID', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
208
- { type: 'FINANCIAL', regex: /\b(?:RENT|LEASE|ESCROW)[\s:]?(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)[0-9,]{3,}\b/gi },
209
- { type: 'SECRET', regex: /\b(?:GATE|DOOR|LOBBY)[-_ ](?:CODE|PIN)[\s:]?\d{4,6}\b/gi }
237
+ { type: 'FINANCIAL', regex: /\b(?:RENT|LEASE|ESCROW)[\s:]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)[0-9,]{3,}\b/gi },
238
+ { type: 'SECRET', regex: /\b(?:GATE|DOOR|LOBBY)[-_ ](?:CODE|PIN)[\s:]*\d{4,6}\b/gi }
210
239
  ],
211
240
  compliance: [
212
241
  { type: 'SECRET', regex: /\b(?:GDPR|HIPAA|CCPA|SOC2|ISO27001)[-_: ]?AUDIT[-_: ]?\d{4}\b/gi },
@@ -214,10 +243,10 @@ let PROFILE_RULES = {
214
243
  { type: 'ID', regex: /\b(?:SAR|DSAR)[-_\/: ]?[A-Z0-9-/]+\b/gi }
215
244
  ],
216
245
  ccpa: [
217
- { type: 'ID', regex: /\bDL[ -]?\d{6,12}\b/gi },
246
+ { type: 'ID', regex: /\b(?:DL|DRIVER['’]?S?\s+LICENSE)[:\s#-]*[A-Z0-9]{6,12}\b/gi },
218
247
  { type: 'LOCATION', regex: /\b-?\d{1,3}\.\d{4,6}[° ]?[NSns],\s*-?\d{1,3}\.\d{4,6}[° ]?[EWew]\b/g },
219
- { type: 'PRIVACY', regex: /\b(?:CCPA|CPRA)[-_: ]?OPT[-_ ]OUT\b/gi },
220
- { type: 'ID', regex: /\bACCOUNT[ -]?(?:ID|NUM|NUMBER)[:\s][A-Z0-9]{6,20}\b/gi }
248
+ { type: 'ID', regex: /\b(?:CCPA|CPRA)[-_: ]?OPT[-_ ]OUT\b/gi },
249
+ { type: 'ID', regex: /\bACCOUNT[ -]?(?:ID|NUM|NUMBER)[:\s]+[A-Z0-9]{6,20}\b/gi }
221
250
  ],
222
251
  engineering: [
223
252
  { type: 'SECRET', regex: /(?<=\b(?:DB|POSTGRES|REDIS|MYSQL|AWS|SECRET|PASSWORD|TOKEN|API|KEY)[A-Z0-9_]*\s*[:=]\s*["']?)[A-Za-z0-9_-]{10,}/gi },
@@ -226,7 +255,7 @@ let PROFILE_RULES = {
226
255
  { type: 'ID', regex: /\b[a-z0-9](?:[-a-z0-9]*[a-z0-9])?\.svc\.cluster\.local\b/g }
227
256
  ],
228
257
  agents: [
229
- { type: 'ID', regex: /\b(?:AGENT|VECTOR|EMBEDDING)[-_: ]?[A-Z0-9]{8,}\b/gi },
258
+ { type: 'ID', regex: /\b(?:AGENT|VECTOR|EMBEDDING)[-_: ]?(?:ID[-_: ]?)?[A-Z0-9]{8,}\b/gi },
230
259
  { type: 'ID', regex: /\bTASK[-_: ]?[A-Z0-9]{5,15}\b/gi },
231
260
  { type: 'SECRET', regex: /\b(?:SYS_PROMPT|SYSTEM_PROMPT|OPENAI_API_KEY)[-_: ]?[A-Za-z0-9_-]{10,}\b/gi }
232
261
  ],
@@ -247,7 +276,7 @@ let PROFILE_RULES = {
247
276
  { type: 'SECRET', regex: /\b(?:CONFIG|KUBECONFIG|TFSTATE)[-_: ]?[A-Z0-9]{6,15}\b/gi }
248
277
  ],
249
278
  personal: [
250
- { type: 'ID', regex: /\b(?:DOB|BIRTHDAY)[\s:]*[0-9./-]{6,10}\b/gi },
279
+ { type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
251
280
  { type: 'SECRET', regex: /\b(?:PASSWORD|PWD|SECRET|PIN)[\s:]*[\S]{4,20}\b/gi },
252
281
  { type: 'PHONE', regex: /\b(?:WIFE|HUSBAND|PARTNER|MOM|DAD)[\s:]+(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}\b/gi }
253
282
  ],
@@ -294,10 +323,46 @@ let PROFILE_RULES = {
294
323
  { type: 'ID', regex: /\b(?:Batch|Lot|Serial)\s*(?:No[.]?|Number|#)?[:\s#]*[A-Z0-9][A-Z0-9\-]{3,14}\b/gi },
295
324
  { type: 'PHI', regex: /\b(?:Dose|Dosage)[:\s]+\d+(?:\.\d+)?\s*(?:mg|mcg|mL|IU|units?)\b/gi },
296
325
  { type: 'ID', regex: /\b(?:CRF|eCRF|Case\s+Report\s+Form)\s*(?:No|Page|ID)?[:\s#]*[A-Z0-9]{2,10}\b/gi }
326
+ ],
327
+ underwriting: [
328
+ // Employer Corporate / Business Names
329
+ { type: 'NAME', isContextName: true, regex: /\b(?:[A-Z][A-Za-z0-9&.,'-]*[ \t\xA0]+){1,5}(?:Inc\.?|LLC|Corp\.?|Corporation|Ltd\.?|Limited|Co\.?|Company|Group|Holdings|Solutions|Services|Technologies|Logistics|Industries|Capital|Bank|Partners|LLP|PLLC)(?:\s+(?:LLC|Inc\.?|Corp\.?|Ltd\.?|USA|Group))?\b/g },
330
+ { type: 'NAME', isContextName: true, regex: /(?:Employer|Company|Organization|Business)\s*(?:Name)?[\s:#]+([A-Za-z0-9&.,\s'-]{2,40}?)(?=\r?\n|$|\s{3,}|\t|Address|EIN|FEIN|Phone|W-2|Rate|Pay|Wage)/gi },
331
+
332
+ // W-2 & Tax Identifiers
333
+ { type: 'ID', regex: /(?:\b(?:Box\s+d\b|d\.\s*(?:Control|#)?|d\s+Control)\s*(?:number|no\.?|#|num)?[:\s#]*|\bControl\s*(?:number|no\.?|#|num)[:\s#]*|\bControl[:#]\s*)([A-Za-z0-9-]{3,30})/gi },
334
+ { type: 'ID', regex: /\b(?:EIN|FEIN|Tax\s+ID)[:\s#]*\d{2}-\d{7}\b/gi },
335
+ { type: 'ID', regex: /\b\d{2}-\d{7}\b/g },
336
+ { type: 'ID', regex: /\b\d{3}-\d{2}-\d{4}\b/g },
337
+ { type: 'ID', regex: /\b(?:XXX|xxx|\*\*\*)[ -]?(?:XX|xx|\*\*)[ -]?\d{4}\b/g },
338
+
339
+ // Employee, Loan & Payroll IDs
340
+ { type: 'ID', regex: /\b(?:Employee|Emp|EE|Worker|Borrower|Badge|Advisor|Producer|Agent|Applicant|File)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Z0-9-]{3,20}\b/gi },
341
+ { type: 'ID', regex: /\b(?:Pay\s+Group|Cost\s+Center|Dept|Department)[:\s#]*[A-Za-z0-9_-]{2,30}\b/gi },
342
+ { type: 'ID', regex: /\b(?:Loan|Application|Deal|Borrower|File)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Za-z0-9-]{4,25}\b/gi },
343
+
344
+ // Direct Deposit, Bank Accounts & Routing Numbers (Masked & Unmasked)
345
+ { type: 'FINANCIAL', regex: /\b(?:Account|Acct|Checking|Savings|Direct\s+Deposit)\s*(?:#|ID|No\.?|Number)?[:\s#]*(?:[\*xX•.-]{3,}\d{2,6}|\d{4}[-\s]?\d{4}[-\s]?\d{2,6})\b/gi },
346
+ { type: 'FINANCIAL', regex: /\b(?:ABA|Routing|RTN)\s*(?:#|ID|No\.?|Number)?[:\s#]*\d{9}\b/gi },
347
+
348
+ // Borrower & Co-Borrower Names (ALL-CAPS, Payroll, Title Case)
349
+ { type: 'NAME', regex: /\b[A-Z]{2,25},\s+[A-Z]{2,25}(?:\s+[A-Z]\.?|\s+[A-Z]{2,25})*\b/g },
350
+ { type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+(?:\p{Lu}\.?|[A-Z][a-z]+))?[ \t\xA0]+\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
351
+ { type: 'NAME', isContextName: true, regex: /(?:(?:Borrower|Co-Borrower|Applicant|Co-Applicant|Employee|Worker|Taxpayer|Candidate|Primary\s+Borrower|Joint\s+Borrower|Account\s+Holder|Insured|Client)\s*(?:Name)?|First\s+name|Given\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9&.,\s'-]{2,35}?)(?=\r?\n|$|\s{3,}|\t|SSN|EIN|DOB|Address|Phone|Rate|Pay|Wage|Date|Box|Last|Surname)/gi },
352
+ { type: 'NAME', isContextName: true, regex: /(?:Last\s+name|Surname|Family\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z'-]{2,30})/gi },
353
+
354
+ // Addresses & Locations
355
+ { type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:Borrower(?:'s)?|Co-Borrower(?:'s)?|Employee(?:'s)?|Employer(?:'s)?|Home|Mailing|Property|Physical)\s+address|(?:(?:\bBox\s+f\b|\bf\.\s*|\bf\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?address))[\s:#]+([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|City|State|ZIP|SSN|EIN|Phone|Box|\d+\b)/gi },
356
+ { type: 'ADDRESS', regex: /\b\d{1,6}[ \t\xA0]+(?:[A-Za-z0-9.-]+[ \t\xA0]+){1,4}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Hwy|Highway|Cir|Circle|Trl|Trail|Loop|Row|Pike|Box|PO Box|P\.O\.[ \t\xA0]*Box)\b(?:[ \t\xA0]*,?[ \t\xA0]*(?:Apt|Apartment|Suite|Ste|Unit|#|Fl|Floor|Bldg|Building)\.?[ \t\xA0]*[A-Za-z0-9-]+)?/gi },
357
+ { type: 'ADDRESS', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
358
+ { type: 'ADDRESS', regex: /\b\d{5}-\d{4}\b/g },
359
+ { type: 'LOCATION', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\b/g }
297
360
  ]
298
361
  };
299
362
 
300
363
  let NAME_STOP_LIST = new Set([
364
+ 'clinical note', 'case note', 'prod log', 'siem alert', 'hr review', 'crm export', 'bank statement', 'file export', 'database row', 'lease application', 'strategy export', 'action log', 'glossary query', 'tool comparison', 'call transcript', 'zendesk ticket', 'board minutes', 'agent context', 'config dump', 'database dump', 'patient note', 'medical record', 'admission note', 'discharge summary', 'progress note', 'hiring review', 'security audit', 'incident response', 'server log', 'system log', 'api response', 'error log', 'audit log', 'debug log',
365
+ 'tax statement', 'wage and tax statement', 'wage and tax', 'wage statement', 'earning statement', 'earnings statement', 'pay statement', 'pay stub', 'paystub', 'withholding statement',
301
366
  'case no', 'account no', 'client no', 'ref no', 'matter no',
302
367
  'affected user', 'incident date', 'incident type', 'incident report',
303
368
  'review period', 'review date', 'salary band', 'salary range',
@@ -339,7 +404,9 @@ let NAME_STOP_LIST = new Set([
339
404
  'step 1', 'step 2', 'step 3', 'step 4', 'step 5',
340
405
  'page 1', 'page 2', 'page 3', 'page 4', 'page 5',
341
406
  'cs101', 'course cs101',
342
- // Expanded Stop List (Common nouns that look like names)
407
+ // Expanded Stop List (Common nouns, command phrases, legal, prompt, chess, and animation terms)
408
+ 'docket number', 'docket numbers', 'dockets section', 'case name', 'case names', 'case number', 'case numbers', 'law firm', 'law firms', 'counsel stack', 'counselstack', 'counselstack connector', 'tier 0', 'tier 1', 'tier 2', 'tier 3', 'tier 4', 'do not', 'do not write', 'specific permission', 'write again', 'without permission', 'without specific permission', 'on screen', 'in report', 'own line', 'connector access', 'prompt instruction', 'prompt instructions', 'finding report', 'findings report',
409
+ 'white bishop', 'black bishop', 'white knight', 'black knight', 'white king', 'black king', 'white queen', 'black queen', 'white rook', 'black rook', 'white pawn', 'black pawn', 'chess piece', 'chess pieces', 'chess game', 'chess match', 'disney-pixar', 'disney pixar', 'pixar animation', 'close-up', 'close up',
343
410
  'january', 'february', 'march', 'april', 'may', 'june', 'july', 'august', 'september', 'october', 'november', 'december',
344
411
  'monday', 'tuesday', 'wednesday', 'thursday', 'friday', 'saturday', 'sunday',
345
412
  'yesterday', 'tomorrow', 'today', 'last week', 'next month', 'early morning', 'late night',
@@ -373,24 +440,104 @@ let NAME_STOP_LIST = new Set([
373
440
  'class name', 'function name', 'variable name', 'database table', 'schema name', 'index name', 'query result', 'error message', 'warning message', 'log entry', 'debug log', 'stack trace',
374
441
  'staff member', 'team member', 'board member', 'board meeting', 'committee member', 'executive board',
375
442
  'email us', 'contact us', 'about us', 'sign in', 'sign out',
443
+ 'driver license', 'drivers license', 'opt out', 'opt-out', 'ccpa opt', 'cpra opt', 'spoiler unreleased', 'unreleased draft',
376
444
  'lighting', 'keyboard', 'creating', 'building', 'training', 'planning', 'starting', 'painting', 'printing', 'returned', 'released', 'required', 'accepted', 'imported', 'services', 'products', 'accounts', 'settings', 'partners', 'keywords', 'keystone', 'keyspace', 'keynotes', 'keychain'
377
- , 'quarterly results', 'strategic planning', 'market research', 'customer base', 'privacy settings', 'account settings', 'security settings', 'download now', 'free trial', 'limited time', 'copyright protected', 'all rights', 'rights reserved', 'credit score', 'monthly rent', 'lease application', 'property address', 'reference number', 'additional identifier', 'lease agreement', 'hiring review', 'candidate name', 'privacy policy', 'terms of service', 'machine learning', 'artificial intelligence', 'generative ai', 'silicon valley', 'google cloud', 'amazon web', 'data science', 'operating system', 'software engineer', 'product manager', 'project manager', 'data analyst', 'gross margin', 'revenue growth', 'source code', 'version control', 'large language model']);
445
+ , 'quarterly results', 'strategic planning', 'market research', 'customer base', 'privacy settings', 'account settings', 'security settings', 'download now', 'free trial', 'limited time', 'copyright protected', 'all rights', 'rights reserved', 'credit score', 'monthly rent', 'lease application', 'property address', 'reference number', 'additional identifier', 'lease agreement', 'hiring review', 'candidate name', 'privacy policy', 'terms of service', 'machine learning', 'artificial intelligence', 'generative ai', 'silicon valley', 'google cloud', 'amazon web', 'data science', 'operating system', 'software engineer', 'product manager', 'project manager', 'data analyst', 'gross margin', 'revenue growth', 'source code', 'version control', 'large language model',
446
+ 'wages', 'wage', 'tips', 'compensation', 'withheld', 'withholding', 'medicare', 'deductions', 'deduction',
447
+ 'regular', 'hours', 'holiday', 'overtime', 'commission', 'bonus', 'bonuses', 'records', 'record', 'statement', 'statements', 'rate', 'rates', 'current', 'ytd', 'benefits', 'taxable', 'pre-tax', 'post-tax', 'reimbursements', 'reimbursement', 'fica', 'oasdi', 'disability', 'unemployment', 'sui', 'sdi', 'std', 'ltd', 'exemptions', 'exemption', 'allowances', 'allowance', 'filing', 'status', 'single', 'married', 'head', 'household', 'advice', 'frequency', 'bi-weekly', 'biweekly', 'weekly', 'monthly', 'semi-monthly', 'direct', 'deposit', 'routing', 'box', 'boxes', 'code', 'control', 'omb', 'copy', 'instructions', 'information', 'deferred', 'adoption', 'statutory', 'third-party', 'sick', 'form', 'schedule', 'w-2', 'w2', 'w-4', 'w4', '1099', 'k-1', '1040', 'fed', 'med', 'fwt', 'swt', 'fed w/h', 'fed med', 'locality', 'state wages', 'state tax', 'local wages', 'local tax', 'allocated', 'nonqualified', 'suff', 'suffix', 'allocated tips', 'advance eic', 'advance eic payment', 'dependent care', 'dependent care benefits', 'nonqualified plans', 'statutory employee', 'retirement plan', 'third-party sick pay']);
378
448
 
379
- let JARGON_WORDS = new Set(['step', 'page', 'grade', 'version', 'course', 'class', 'follow', 'chapter', 'lesson', 'unit', 'marketing', 'manager', 'specialist', 'science', 'administration', 'university', 'skills', 'leadership', 'communication', 'working', 'proficiency', 'decision', 'driven', 'experience', 'summary', 'bachelor', 'ads', 'solutions', 'positioning', 'acquisition', 'strategy', 'research', 'database', 'forecast', 'interest', 'prepared', 'merchant', 'document', 'feedback', 'template', 'campaign', 'partners', 'settings', 'keystone', 'llm', 'gpt', 'chatgpt', 'openai', 'anthropic', 'claude', 'gemini', 'api', 'json', 'xml', 'html', 'css', 'javascript', 'python', 'policy', 'terms', 'conditions', 'release', 'sprint', 'deployment', 'cluster', 'instance', 'package', 'module', 'revenue', 'margin', 'gross', 'quarter', 'system', 'code', 'data', 'cloud', 'server', 'database', 'artificial', 'intelligence', 'learning', 'generative']);
449
+ let JARGON_WORDS = new Set(['step', 'page', 'grade', 'version', 'course', 'class', 'follow', 'chapter', 'lesson', 'unit', 'marketing', 'manager', 'specialist', 'science', 'administration', 'university', 'skills', 'leadership', 'communication', 'working', 'proficiency', 'decision', 'driven', 'experience', 'summary', 'bachelor', 'ads', 'solutions', 'positioning', 'acquisition', 'strategy', 'research', 'database', 'forecast', 'interest', 'prepared', 'merchant', 'document', 'feedback', 'template', 'campaign', 'partners', 'settings', 'keystone', 'llm', 'gpt', 'chatgpt', 'openai', 'anthropic', 'claude', 'gemini', 'api', 'json', 'xml', 'html', 'css', 'javascript', 'python', 'golang', 'typescript', 'rust', 'fastapi', 'snowflake', 'kubernetes', 'terraform', 'docker', 'redis', 'kafka', 'pytorch', 'policy', 'terms', 'conditions', 'release', 'sprint', 'deployment', 'cluster', 'instance', 'package', 'module', 'revenue', 'margin', 'gross', 'quarter', 'system', 'code', 'data', 'cloud', 'server', 'database', 'artificial', 'intelligence', 'learning', 'generative', 'regular', 'hours', 'holiday', 'earnings', 'deductions', 'withheld', 'withholding', 'taxes', 'medicare', 'benefits', 'reimbursements', 'compensation', 'wages', 'code']);
380
450
 
381
451
  let NOT_NAME_WORDS = new Set([
382
452
  // Grammatical & Sentence Starters
383
- 'the', 'a', 'an', 'this', 'that', 'these', 'those', 'my', 'your', 'his', 'her', 'their', 'our', 'its', 'it', 'he', 'she', 'they', 'we', 'i', 'you', 'who', 'whom', 'which', 'what', 'whose', 'why', 'how', 'when', 'where', 'with', 'for', 'from', 'by', 'to', 'at', 'in', 'on', 'of', 'about', 'as', 'into', 'through', 'during', 'before', 'after', 'above', 'below', 'and', 'but', 'or', 'so', 'yet',
453
+ 'the', 'a', 'an', 'this', 'that', 'these', 'those', 'my', 'your', 'his', 'her', 'their', 'our', 'its', 'it', 'he', 'she', 'they', 'we', 'i', 'you', 'who', 'whom', 'which', 'what', 'whose', 'why', 'how', 'when', 'where', 'with', 'for', 'from', 'by', 'to', 'at', 'in', 'on', 'of', 'about', 'as', 'into', 'through', 'during', 'before', 'after', 'above', 'below', 'and', 'but', 'or', 'so', 'yet', 'im', "i'm", "you're", "they're", "we're", "it's", "he's", "she's", "that's", "there's", "what's", "who's", "i've", "you've", "we've", "they've", "i'll", "you'll", "we'll", "they'll", "i'd", "you'd", "we'd", "they'd",
454
+ // Verbs, Auxiliaries, Commands & Imperatives
455
+ 'do', 'does', 'did', 'done', 'doing', 'dont', "don't", 'doesnt', "doesn't", 'didnt', "didn't", 'not', 'no', 'never', 'always',
456
+ 'be', 'is', 'am', 'are', 'was', 'were', 'been', 'being',
457
+ 'have', 'has', 'had', 'having',
458
+ 'can', 'could', 'may', 'might', 'must', 'shall', 'should', 'will', 'would', 'wont', "won't", 'wouldnt', "wouldn't", 'shouldnt', "shouldn't", 'couldnt', "couldn't", 'cant', "can't", 'cannot',
459
+ 'write', 'writing', 'written', 'writes', 'read', 'reading', 'reads',
460
+ 'wait', 'waiting', 'waited', 'waits', 'place', 'placing', 'placed', 'places',
461
+ 'display', 'displaying', 'displayed', 'displays',
462
+ 'provide', 'providing', 'provided', 'provides',
463
+ 'show', 'showing', 'shown', 'shows',
464
+ 'tell', 'telling', 'told', 'tells',
465
+ 'ask', 'asking', 'asked', 'asks',
466
+ 'use', 'using', 'used', 'uses',
467
+ 'select', 'selecting', 'selected', 'selects',
468
+ 'find', 'finding', 'findings', 'found', 'finds',
469
+ 'reference', 'referencing', 'referenced', 'references',
470
+ 'access', 'accessing', 'accessed', 'accesses',
471
+ 'note', 'noting', 'noted', 'notes',
472
+ 'get', 'getting', 'got', 'gotten', 'gets',
473
+ 'make', 'making', 'made', 'makes',
474
+ 'give', 'giving', 'given', 'gives',
475
+ 'take', 'taking', 'took', 'taken', 'takes',
476
+ 'put', 'putting', 'puts',
477
+ 'set', 'setting', 'sets',
478
+ 'keep', 'keeping', 'kept', 'keeps',
479
+ 'let', 'letting', 'lets',
480
+ 'leave', 'leaving', 'left', 'leaves',
481
+ 'run', 'running', 'ran', 'runs',
482
+ 'stop', 'stopping', 'stopped', 'stops',
483
+ 'start', 'starting', 'started', 'starts',
484
+ 'check', 'checking', 'checked', 'checks',
485
+ 'print', 'printing', 'printed', 'prints',
486
+ 'generate', 'generating', 'generated', 'generates',
487
+ 'create', 'creating', 'created', 'creates',
488
+ 'build', 'building', 'built', 'builds',
489
+ 'include', 'including', 'included', 'includes',
490
+ 'exclude', 'excluding', 'excluded', 'excludes',
491
+ 'format', 'formatting', 'formatted', 'formats',
492
+ 'change', 'changing', 'changed', 'changes',
493
+ 'send', 'sending', 'sent', 'sends',
494
+ 'receive', 'receiving', 'received', 'receives',
495
+ 'delete', 'deleting', 'deleted', 'deletes',
496
+ 'remove', 'removing', 'removed', 'removes',
497
+ 'insert', 'inserting', 'inserted', 'inserts',
498
+ 'update', 'updating', 'updated', 'updates',
499
+ 'review', 'reviewing', 'reviewed', 'reviews',
500
+ 'allow', 'allowing', 'allowed', 'allows',
501
+ 'deny', 'denying', 'denied', 'denies',
502
+ 'require', 'requiring', 'required', 'requires',
503
+ 'turn', 'turning', 'turned', 'turns',
504
+ 'switch', 'switching', 'switched', 'switches',
505
+ 'enable', 'enabling', 'enabled', 'enables',
506
+ 'disable', 'disabling', 'disabled', 'disables',
507
+ 'ensure', 'ensuring', 'ensured', 'ensures',
508
+ 'verify', 'verifying', 'verified', 'verifies',
509
+ 'execute', 'executing', 'executed', 'executes',
510
+ 'test', 'testing', 'tested', 'tests',
511
+ 'install', 'installing', 'installed', 'installs',
512
+ 'uninstall', 'uninstalling', 'uninstalled', 'uninstalls',
513
+ 'suppose', 'supposed', 'supposing', 'supposes',
514
+ 'respond', 'responding', 'responded', 'responds',
515
+ 'preserve', 'preserving', 'preserved', 'preserves',
516
+ 'replace', 'replacing', 'replaced', 'replaces',
517
+ // Adverbs, Prepositions, Conjunctions & Modifiers
518
+ 'again', 'without', 'with', 'within', 'specific', 'specifically', 'permission', 'permissions',
519
+ 'underneath', 'above', 'below', 'between', 'among', 'together', 'separately', 'instead',
520
+ 'also', 'too', 'either', 'neither', 'both', 'each', 'every', 'all', 'some', 'any', 'none',
521
+ 'only', 'just', 'already', 'currently', 'more', 'most', 'less', 'least',
522
+ 'very', 'quite', 'rather', 'such', 'same', 'different', 'other', 'others', 'another',
523
+ 'like', 'unlike', 'similar', 'complete', 'completely', 'entire', 'entirely',
524
+ 'exact', 'exactly', 'approximate', 'approximately', 'general', 'generally',
525
+ 'direct', 'directly', 'indirect', 'indirectly', 'total', 'totally', 'full', 'fully',
526
+ 'partial', 'partially', 'own', 'proper', 'properly',
527
+ 'now', 'then', 'soon', 'later', 'here', 'there', 'everywhere', 'nowhere', 'somewhere', 'anywhere',
528
+ 'inside', 'outside', 'before', 'after', 'since', 'until', 'till',
529
+ 'while', 'whereas', 'unless', 'although', 'though', 'even', 'because',
530
+ 'therefore', 'however', 'furthermore', 'moreover', 'meanwhile', 'otherwise', 'besides', 'further',
384
531
  // Greetings & Salutations
385
532
  'hello', 'hi', 'hey', 'dear', 'greetings',
386
533
  // Document & Resume Structure
387
- 'summary', 'experience', 'education', 'skills', 'languages', 'project', 'history', 'background', 'objective', 'profile', 'awards', 'honors', 'certifications', 'publications', 'interests', 'references',
534
+ 'summary', 'experience', 'education', 'skills', 'languages', 'project', 'history', 'background', 'objective', 'profile', 'awards', 'honors', 'certifications', 'publications', 'interests', 'references', 'statement', 'statements', 'form', 'forms',
388
535
  // Business & Job Roles
389
536
  'manager', 'director', 'specialist', 'analyst', 'engineer', 'developer', 'consultant', 'officer', 'representative', 'agent', 'lead', 'leader', 'president', 'coordinator', 'admin', 'administrator', 'executive', 'founder', 'partner', 'intern', 'trainee', 'advisor', 'head', 'vp', 'chief',
390
537
  // Departments & Fields
391
538
  'marketing', 'sales', 'engineering', 'finance', 'accounting', 'legal', 'operations', 'support', 'recruiting', 'talent', 'acquisition', 'compliance', 'security', 'technical', 'development', 'product', 'design', 'creative', 'strategy', 'planning', 'analytics', 'science', 'business', 'administration',
392
539
  // Tools & Tech Concepts
393
- 'google', 'ads', 'analytics', 'meta', 'hubspot', 'crm', 'salesforce', 'wordpress', 'mailchimp', 'adobe', 'figma', 'canva', 'slack', 'zoom', 'teams', 'microsoft', 'office', 'excel', 'word', 'powerpoint', 'notion', 'jira', 'confluence', 'github', 'gitlab', 'aws', 'azure', 'cloud', 'database', 'sql', 'python', 'java', 'javascript', 'html', 'css', 'react', 'node', 'api', 'saas', 'b2b', 'b2c', 'url', 'domain', 'website', 'app', 'application', 'software', 'email', 'phone', 'contact', 'address',
540
+ 'google', 'ads', 'analytics', 'meta', 'hubspot', 'crm', 'salesforce', 'wordpress', 'mailchimp', 'adobe', 'figma', 'canva', 'slack', 'zoom', 'teams', 'microsoft', 'office', 'excel', 'word', 'powerpoint', 'notion', 'jira', 'confluence', 'github', 'gitlab', 'aws', 'gcp', 'azure', 'cloud', 'database', 'sql', 'python', 'golang', 'typescript', 'rust', 'fastapi', 'snowflake', 'kubernetes', 'terraform', 'docker', 'redis', 'kafka', 'pytorch', 'java', 'javascript', 'html', 'css', 'react', 'node', 'api', 'saas', 'b2b', 'b2c', 'url', 'domain', 'website', 'app', 'application', 'software', 'email', 'phone', 'contact', 'address',
394
541
 
395
542
  // General Academic & Professional vocabulary
396
543
  'bachelor', 'master', 'doctor', 'associate', 'degree', 'university', 'college', 'school', 'institute', 'academy', 'graduated', 'major', 'minor', 'gpa', 'cum', 'laude', 'honors', 'deans', 'list', 'scholarship',
@@ -399,15 +546,29 @@ let NOT_NAME_WORDS = new Set([
399
546
  // Common Resume / Business Phrases
400
547
  'results-driven', 'data-driven', 'customer-centric', 'detail-oriented', 'cross-functional', 'self-motivated', 'time-management', 'problem-solving', 'fast-paced', 'year-over-year',
401
548
  // Legal & Trust terms
402
- 'trust', 'trustee', 'co-trustee', 'settlor', 'grantor', 'beneficiary', 'agreement', 'will', 'estate', 'witness', 'declaration', 'signatory', 'testator', 'notary', 'commission', 'county', 'state', 'court', 'article', 'section', 'paragraph', 'schedule', 'exhibit', 'amendment', 'addendum', 'power', 'attorney', 'guardian', 'executor', 'administrator', 'survivor', 'predecessor', 'successor', 'whereof', 'hereby', 'thereby', 'herein', 'therein', 'witnesseth', 'whereas', 'therefore', 'now', 'dated', 'effective',
549
+ 'trust', 'trustee', 'co-trustee', 'settlor', 'grantor', 'beneficiary', 'agreement', 'will', 'estate', 'witness', 'declaration', 'signatory', 'testator', 'notary', 'commission', 'county', 'state', 'court', 'article', 'section', 'paragraph', 'schedule', 'exhibit', 'amendment', 'addendum', 'power', 'attorney', 'guardian', 'executor', 'administrator', 'survivor', 'predecessor', 'successor', 'whereof', 'hereby', 'thereby', 'herein', 'therein', 'witnesseth', 'whereas', 'therefore', 'now', 'dated', 'effective', 'matter', 'case', 'cases', 'docket', 'dockets', 'number', 'numbers', 'firm', 'firms', 'lawyer', 'lawyers', 'counsel', 'counsels', 'counselstack', 'tier', 'tiers', 'finding', 'findings', 'connector', 'connectors', 'platform', 'platforms',
550
+ // Medical & Clinical terms
551
+ 'clinical', 'note', 'notes', 'dx', 'rx', 'tx', 'hx', 'px', 'sx', 'type', 'diabetes', 'referred', 'referral', 'diagnosed', 'diagnosis', 'patient', 'insurance', 'bcbs', 'mrn', 'dob',
552
+ // Tax & Payroll terms
553
+ 'wages', 'wage', 'tips', 'compensation', 'withheld', 'withholding', 'medicare', 'deductions', 'deduction', 'earning', 'earnings', 'gross', 'net', 'pay', 'payroll', 'paystub', 'taxable', 'exempt', 'allowance', 'allowances', 'regular', 'hours', 'holiday', 'overtime', 'commission', 'bonus', 'bonuses', 'records', 'record', 'statement', 'statements', 'rate', 'rates', 'current', 'ytd', 'benefits', 'taxable', 'pre-tax', 'post-tax', 'reimbursements', 'reimbursement', 'fica', 'oasdi', 'disability', 'unemployment', 'sui', 'sdi', 'std', 'ltd', 'exemptions', 'exemption', 'allowances', 'allowance', 'filing', 'status', 'single', 'married', 'head', 'household', 'advice', 'frequency', 'bi-weekly', 'biweekly', 'weekly', 'monthly', 'semi-monthly', 'direct', 'deposit', 'routing', 'box', 'boxes', 'code', 'control', 'omb', 'copy', 'instructions', 'information', 'deferred', 'adoption', 'statutory', 'third-party', 'sick', 'form', 'schedule', 'w-2', 'w2', 'w-4', 'w4', '1099', 'k-1', '1040', 'fed', 'med', 'fwt', 'swt', 'fed w/h', 'fed med', 'locality', 'state wages', 'state tax', 'local wages', 'local tax', 'allocated', 'nonqualified',
554
+ // Common Web, UI, Compliance, Document & AI Terms (Suppresses false-positive Name detection on headlines, buttons, and badges)
555
+ 'types', 'type', 'leaked', 'leak', 'leaks', 'masked', 'mask', 'masking', 'leave', 'screen', 'screens', 'risk', 'risks', 'cluster', 'clusters', 'parameter', 'parameters', 'processing', 'process', 'processed', 'verified', 'verify', 'verification', 'playground', 'guide', 'guides', 'protection', 'protect', 'corporate', 'enterprise', 'log', 'logs', 'airplane', 'mode', 'zero', 'trust', 'top', 'data', 'live', 'scrubber', 'scrub', 'scrubbed', 'note', 'notes', 'secret', 'secrets', 'card', 'cards', 'raw', 'input', 'output', 'contains', 'contain', 'contained', 'platform', 'solutions', 'pricing', 'company', 'news', 'dashboard', 'add', 'chrome', 'sample', 'samples', 'try', 'terms', 'privacy', 'policy', 'policies', 'home', 'compliance', 'framework', 'frameworks', 'audit', 'audits', 'receipt', 'receipts', 'overview', 'explore', 'vectors', 'vector', 'standard', 'standards', 'status', 'preview', 'view', 'actions', 'action', 'button', 'buttons', 'option', 'options', 'general', 'specialized', 'custom', 'rule', 'rules', 'token', 'tokens', 'value', 'values', 'session', 'sessions', 'local', 'server', 'servers', 'cloud', 'ram', 'memory', 'offline', 'online', 'client', 'browser', 'extension', 'workspace', 'workplace', 'pan', 'phi', 'pii', 'soc', 'soc2', 'gdpr', 'hipaa', 'ccpa', 'iso27001', 'pci', 'dss', 'nist', 'chatgpt', 'claude', 'gemini', 'copilot', 'perplexity', 'deepseek', 'qwen', 'grok', 'llama', 'mistral', 'ai', 'llm', 'prompt', 'prompts', 'transmission', 'transit', 'egress', 'neutralized', 'stripped', 'isolated', 'isolation', 'unlocked', 'locked', 'unlock', 'download', 'copy', 'dismiss', 'close', 'save', 'settings', 'protect', 'reveal', 'unmask', 'restore', 'restored', 'export', 'import',
556
+ // Games, Chess, and Playing Pieces
557
+ 'bishop', 'bishops', 'knight', 'knights', 'rook', 'rooks', 'pawn', 'pawns', 'king', 'kings', 'queen', 'queens', 'chessboard', 'checkmate', 'stalemate', 'castling', 'en passant', 'chess',
558
+ // Colors & Visual Descriptors
559
+ 'white', 'black', 'red', 'blue', 'green', 'yellow', 'orange', 'purple', 'pink', 'brown', 'gray', 'grey', 'dark', 'light', 'gold', 'silver', 'bronze',
560
+ // Animation, 3D Rendering & Prompt Terminology
561
+ 'pixar', 'disney', 'animation', 'render', 'rendering', 'composition', 'cinematic', 'smooth', 'glides', 'glide', 'gliding', 'capture', 'captures', 'capturing', 'camera', 'orbit', 'orbits', 'orbiting', 'trapped', 'trap', 'trapping', 'square', 'squares', 'character', 'characters', 'expressive', 'living', 'texture', 'textures', 'reflection', 'reflections', 'grain', 'candlelight', 'wooden', 'polished', 'vertical', 'horizontal', 'macro', 'closeup', 'close-up', 'scene', 'scenes', 'shot', 'shots', 'shadow', 'shadows',
562
+ // Email, Outreach, Guest Posting & Agency Business Vocabulary
563
+ 'guest', 'post', 'posts', 'posting', 'attached', 'attach', 'attachment', 'attachments', 'updated', 'update', 'updates', 'list', 'lists', 'line', 'lines', 'rate', 'rates', 'affordable', 'services', 'service', 'infotech', 'technologies', 'technology', 'agency', 'agencies', 'digital', 'marketing', 'traffic', 'smart', 'design', 'seo', 'per', 'host', 'hosting', 'sites', 'site', 'inbox', 'starred', 'snoozed', 'important', 'sent', 'drafts', 'draft', 'spam', 'bin', 'trash', 'purchases', 'travel', 'social', 'forums', 'promotions', 'promotion', 'reply', 'forward', 'labels', 'label', 'compose', 'message', 'messages', 'mailer', 'outreach', 'backlink', 'backlinks', 'domain', 'authority', 'da', 'dr', 'founder', 'ceo', 'cto', 'cfo', 'coo', 'vp', 'head', 'lead',
403
564
  // US States
404
- 'california', 'texas', 'florida', 'york', 'illinois', 'pennsylvania', 'ohio', 'georgia', 'michigan', 'carolina', 'virginia', 'washington', 'arizona', 'massachusetts', 'tennessee', 'indiana', 'maryland', 'missouri', 'wisconsin', 'colorado', 'minnesota', 'alabama', 'louisiana', 'kentucky', 'oregon', 'oklahoma', 'connecticut', 'utah', 'iowa', 'nevada', 'arkansas', 'mississippi', 'kansas', 'new mexico', 'nebraska', 'idaho', 'hawaii', 'maine', 'new hampshire', 'rhode island', 'montana', 'delaware', 'south dakota', 'north dakota', 'alaska', 'vermont', 'wyoming'
405
- , 'f.3d', 'f.supp', 'u.s.c.', 'v.', 'plaintiff', 'defendant', 'v', 'u.s.', 'court', 'app.', 'reporter', 'cir.']);
565
+ 'california', 'texas', 'florida', 'york', 'illinois', 'pennsylvania', 'ohio', 'georgia', 'michigan', 'carolina', 'virginia', 'washington', 'arizona', 'massachusetts', 'tennessee', 'indiana', 'maryland', 'missouri', 'wisconsin', 'colorado', 'minnesota', 'alabama', 'louisiana', 'kentucky', 'oregon', 'oklahoma', 'connecticut', 'utah', 'iowa', 'nevada', 'arkansas', 'mississippi', 'kansas', 'new mexico', 'nebraska', 'idaho', 'hawaii', 'maine', 'new hampshire', 'rhode island', 'montana', 'delaware', 'south dakota', 'north dakota', 'alaska', 'vermont', 'wyoming',
566
+ 'f.3d', 'f.supp', 'u.s.c.', 'v.', 'plaintiff', 'defendant', 'v', 'u.s.', 'court', 'app.', 'reporter', 'cir.']);
406
567
 
407
568
  const PROFILE_JARGON = {
408
- medical: ['sleep', 'apnea', 'symptom', 'symptoms', 'trauma', 'hypertension', 'health', 'disease', 'condition', 'diagnosis', 'treatment', 'medication', 'dose', 'patient', 'clinic', 'surgery', 'therapy', 'alcohol', 'cannabis', 'blood', 'pressure', 'heart', 'rate', 'emergency', 'contact', 'relationship'],
569
+ medical: ['sleep', 'apnea', 'symptom', 'symptoms', 'trauma', 'hypertension', 'health', 'disease', 'condition', 'diagnosis', 'treatment', 'medication', 'dose', 'patient', 'clinic', 'surgery', 'therapy', 'alcohol', 'cannabis', 'blood', 'pressure', 'heart', 'rate', 'emergency', 'contact', 'relationship', 'type', 'diabetes', 'cancer', 'asthma', 'copd', 'covid', 'infection', 'syndrome', 'disorder', 'chronic', 'acute', 'illness', 'fever', 'allergy', 'pain', 'referral', 'referred', 'prescription', 'prescribed', 'doctor', 'physician', 'nurse', 'hospital', 'clinical', 'note', 'notes', 'dx', 'rx', 'tx', 'hx', 'px', 'sx', 'insurance', 'bcbs'],
409
570
  realestate: ['escrow', 'tenant', 'landlord', 'lease', 'mortgage', 'appraisal', 'broker', 'property', 'zoning', 'parcel', 'rent', 'buyer', 'seller', 'agent', 'listing'],
410
- legal: ['testator', 'notary', 'commission', 'county', 'court', 'affidavit', 'plaintiff', 'defendant', 'litigation', 'jurisdiction', 'agreement', 'contract', 'settlement', 'clause', 'article', 'section'],
571
+ legal: ['testator', 'notary', 'commission', 'county', 'court', 'affidavit', 'plaintiff', 'defendant', 'litigation', 'jurisdiction', 'agreement', 'contract', 'settlement', 'clause', 'article', 'section', 'matter', 'case'],
411
572
  hr: ['candidate', 'employee', 'payroll', 'benefits', 'salary', 'vacation', 'supervisor', 'subordinate', 'performance', 'appraisal', 'interview', 'resume', 'applicant'],
412
573
  sales: ['prospect', 'opportunity', 'quota', 'pipeline', 'deal', 'revenue', 'forecast', 'lead', 'churn', 'client', 'customer']
413
574
  };
@@ -447,11 +608,14 @@ const PROFILE_JARGON = {
447
608
  }
448
609
 
449
610
  const PROFILE_ALIAS_MAP = {
611
+ 'general': 'general',
612
+ 'underwriting': 'underwriting', 'lending': 'underwriting', 'mortgage': 'underwriting', 'loan': 'underwriting', 'income': 'underwriting', 'income_verification': 'underwriting', 'payroll': 'underwriting', 'w2': 'underwriting', 'paystub': 'underwriting',
450
613
  'medical': 'medical', 'healthcare': 'medical', 'health': 'medical', 'pharma': 'pharma',
451
- 'engineering': 'engineering', 'dev': 'engineering', 'tech': 'tech',
452
- 'finance': 'finance', 'bizops': 'bizops', 'sales': 'sales', 'wealthmgmt': 'wealthmgmt', 'insurance': 'insurance', 'accounting': 'accounting',
614
+ 'engineering': 'engineering', 'dev': 'engineering', 'devops': 'engineering', 'tech': 'tech',
615
+ 'finance': 'finance', 'bizops': 'bizops', 'sales': 'sales', 'wealthmgmt': 'wealthmgmt', 'wealth': 'wealthmgmt', 'insurance': 'insurance', 'accounting': 'accounting',
453
616
  'legal': 'legal', 'compliance': 'compliance', 'ccpa': 'ccpa',
454
617
  'hr': 'hr', 'security': 'security', 'marketing': 'marketing', 'support': 'support',
618
+ 'realestate': 'realestate', 'academic': 'academic', 'agents': 'agents', 'ai_agents': 'agents', 'creative': 'creative', 'personal': 'personal'
455
619
  };
456
620
 
457
621
  function getActiveRules(activeProfile) {
@@ -459,7 +623,7 @@ const PROFILE_JARGON = {
459
623
  if (activeProfile && activeProfile.toLowerCase() !== 'general') {
460
624
  const canonicalProfile = PROFILE_ALIAS_MAP[activeProfile.toLowerCase()] || 'general';
461
625
  if (canonicalProfile !== 'general' && PROFILE_RULES[canonicalProfile]) {
462
- activeRules = activeRules.concat(PROFILE_RULES[canonicalProfile]);
626
+ activeRules = PROFILE_RULES[canonicalProfile].concat(activeRules);
463
627
  }
464
628
  }
465
629
  return activeRules;
@@ -482,14 +646,15 @@ const PROFILE_JARGON = {
482
646
 
483
647
  if (customRules && customRules.length > 0) {
484
648
  const sorted = [...customRules].sort((a, b) => {
485
- const patternA = typeof a === 'string' ? a : a.pattern;
486
- const patternB = typeof b === 'string' ? b : b.pattern;
649
+ const patternA = typeof a === 'string' ? a : (a.pattern || (a.regex ? a.regex.source : '') || '');
650
+ const patternB = typeof b === 'string' ? b : (b.pattern || (b.regex ? b.regex.source : '') || '');
487
651
  return patternB.length - patternA.length;
488
652
  });
489
653
 
490
654
  sorted.forEach(cr => {
491
- const pattern = typeof cr === 'string' ? cr : cr.pattern;
492
- const label = typeof cr === 'string' ? 'CUSTOM' : (cr.label || 'CUSTOM');
655
+ const pattern = typeof cr === 'string' ? cr : (cr.pattern || (cr.regex ? cr.regex.source : ''));
656
+ if (!pattern) return;
657
+ const label = typeof cr === 'string' ? 'CUSTOM' : (cr.label || cr.mask || cr.name || 'CUSTOM');
493
658
 
494
659
  let rx;
495
660
  try {
@@ -525,13 +690,23 @@ const PROFILE_JARGON = {
525
690
  let matchedText = m[0];
526
691
  let start = m.index;
527
692
 
528
- if (rule.isContextName && m[1]) {
693
+ if (m.length > 1 && m[1] !== undefined && m[1] !== '') {
529
694
  matchedText = m[1];
530
- start = m.index + m[0].lastIndexOf(m[1]);
695
+ const relOffset = m[0].indexOf(m[1]);
696
+ if (relOffset !== -1) {
697
+ start = m.index + relOffset;
698
+ }
531
699
  }
532
700
  const end = start + matchedText.length;
533
701
 
534
702
  if (rule.type !== 'NAME' && rule.type !== 'ADDRESS') {
703
+ const val = matchedText.toLowerCase().trim();
704
+ if (NAME_STOP_LIST.has(val) || NOT_NAME_WORDS.has(val)) {
705
+ // Skip if generic English dictionary term matched by greedy regex (e.g. SWIFT matching SCRUBBER or CONTAINS)
706
+ if (rule.type === 'FINANCIAL' || rule.type === 'ID' || rule.type === 'PRIVACY' || rule.type === 'SECRET') {
707
+ continue;
708
+ }
709
+ }
535
710
  matches.push({ start, end, value: matchedText, type: rule.type });
536
711
  } else {
537
712
  let val = matchedText.toLowerCase().trim();
@@ -540,18 +715,26 @@ const PROFILE_JARGON = {
540
715
  if (!rule.isContextName && rule.type === 'NAME') {
541
716
  let words = val.split(/[ \t\xA0]+/);
542
717
  let origWords = matchedText.split(/[ \t\xA0]+/);
543
- while (words.length > 2 && (currentJargon.has(words[0]) || NOT_NAME_WORDS.has(words[0]))) {
718
+ while (words.length > 2 && (currentJargon.has(words[0]) || (words[0].length > 1 && NOT_NAME_WORDS.has(words[0])) || (words[0].replace(/[^\p{L}]/gu, '').length > 1 && NOT_NAME_WORDS.has(words[0].replace(/[^\p{L}]/gu, ''))))) {
544
719
  origWords.shift();
545
720
  words.shift();
546
721
  const nextStart = matchedText.indexOf(origWords[0]);
547
- start += nextStart;
548
- matchedText = matchedText.substring(nextStart);
549
- val = matchedText.toLowerCase().trim();
722
+ if (nextStart !== -1) {
723
+ start += nextStart;
724
+ matchedText = matchedText.substring(nextStart);
725
+ val = matchedText.toLowerCase().trim();
726
+ } else {
727
+ break;
728
+ }
550
729
  }
551
- if (words.some(w => currentJargon.has(w) || NOT_NAME_WORDS.has(w))) continue;
730
+ if (words.some(w => {
731
+ const cleanW = w.replace(/[^\p{L}]/gu, '');
732
+ if (cleanW.length <= 1) return false;
733
+ return currentJargon.has(w) || NOT_NAME_WORDS.has(w) || NOT_NAME_WORDS.has(cleanW) || NAME_STOP_LIST.has(w) || NAME_STOP_LIST.has(cleanW);
734
+ })) continue;
552
735
  } else if (rule.type === 'ADDRESS') {
553
- const words = val.split(/\s+/);
554
- if (words.some(w => currentJargon.has(w))) continue;
736
+ const cleanVal = val.replace(/[.,;!?]/g, ' ').trim();
737
+ if (NAME_STOP_LIST.has(cleanVal) || currentJargon.has(cleanVal)) continue;
555
738
  }
556
739
 
557
740
  matches.push({ start, end: start + matchedText.length, value: matchedText, type: rule.type });
@@ -588,7 +771,7 @@ const PROFILE_JARGON = {
588
771
  nameWords.forEach(w => {
589
772
  if (w && w.length >= 2 && /^\p{Lu}/u.test(w)) {
590
773
  const wl = w.toLowerCase();
591
- if (!NOT_NAME_WORDS.has(wl) && !JARGON_WORDS.has(wl) && !NAME_STOP_LIST.has(wl)) {
774
+ if (!NOT_NAME_WORDS.has(wl) && !JARGON_WORDS.has(wl) && !NAME_STOP_LIST.has(wl) && !currentJargon.has(wl)) {
592
775
  learnedNames.add(w);
593
776
  }
594
777
  }
@@ -641,13 +824,24 @@ const PROFILE_JARGON = {
641
824
  return Array.from(aliases);
642
825
  }
643
826
 
827
+ function formatToken(label, index, format = 'brackets') {
828
+ const cleanLabel = String(label || 'PII').replace(/[^A-Za-z0-9_]/g, '_').toUpperCase();
829
+ switch(format) {
830
+ case 'xml': return `<${cleanLabel}_${index}>`;
831
+ case 'mustache': return `{{${cleanLabel}_${index}}}`;
832
+ case 'underscores': return `__${cleanLabel}_${index}__`;
833
+ case 'brackets':
834
+ default: return `[${cleanLabel}_${index}]`;
835
+ }
836
+ }
837
+
644
838
  function buildRestorationRegexAndRules(tokenMap) {
645
839
  const ObjectKeys = Object.keys(tokenMap);
646
840
  if (ObjectKeys.length === 0) return { compositeRegex: null, looseRules: [] };
647
841
 
648
842
  const sortedKeys = [...ObjectKeys].sort((a, b) => {
649
- const innerA = a.replace(/^\[|\]$/g, '');
650
- const innerB = b.replace(/^\[|\]$/g, '');
843
+ const innerA = a.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
844
+ const innerB = b.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
651
845
  const matchA = innerA.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
652
846
  const matchB = innerB.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
653
847
  if (matchA && matchB) {
@@ -665,7 +859,7 @@ const PROFILE_JARGON = {
665
859
  const regexParts = [];
666
860
 
667
861
  sortedKeys.forEach(k => {
668
- const inner = k.replace(/^\[|\]$/g, '');
862
+ const inner = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
669
863
  const match = inner.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
670
864
  if (match) {
671
865
  const label = match[1];
@@ -673,7 +867,8 @@ const PROFILE_JARGON = {
673
867
  const aliases = getLabelAliases(label);
674
868
 
675
869
  const escapedAliases = aliases.map(a => a.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
676
- const loosePattern = `\\[\\s*(?:${escapedAliases.join('|')})[-_\\s]*0*${baseIndex}\\s*\\]|(?<![A-Za-z0-9\\u0400-\\u04FF_])(?:${escapedAliases.join('|')})[-_\\s]*0*${baseIndex}(?![A-Za-z0-9\\u0400-\\u04FF_])`;
870
+ const aliasesGroup = `(?:${escapedAliases.join('|')})`;
871
+ const loosePattern = `(?:\\[\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*\\]|<\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*>|\\{\\{\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*\\}\\}|__\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*__|(?<![A-Za-z0-9\\u0400-\\u04FF_])${aliasesGroup}[-_\\s]*0*${baseIndex}(?![A-Za-z0-9\\u0400-\\u04FF_]))(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})?`;
677
872
  looseRules.push({ token: k, pattern: loosePattern });
678
873
  }
679
874
  regexParts.push(k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
@@ -681,42 +876,157 @@ const PROFILE_JARGON = {
681
876
 
682
877
  let compositeRegex = null;
683
878
  if (regexParts.length > 0) {
684
- compositeRegex = new RegExp(`(?:\\b|\\[)?(?:(?:(?<!\\w)|(?<=\\s))(?:${regexParts.join('|')})(?:(?!\\w)|(?=\\s)))(?:\\b|\\])?`, 'g');
879
+ compositeRegex = new RegExp(`(?:\\b|\\[|<|\\{\\{|__)?(?:(?:(?<!\\w)|(?<=\\s))(?:${regexParts.join('|')})(?:(?!\\w)|(?=\\s)))(?:\\b|\\]|>|\\}\\}|__)?(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})?`, 'g');
685
880
  }
686
881
 
687
882
  return { compositeRegex, looseRules };
688
883
  }
689
884
 
885
+ function isJsonPayload(str) {
886
+ if (!str || typeof str !== "string") return false;
887
+ const trimmed = str.trim();
888
+ if (!((trimmed.startsWith("{") && trimmed.endsWith("}")) || (trimmed.startsWith("[") && trimmed.endsWith("]")))) {
889
+ return false;
890
+ }
891
+ try {
892
+ JSON.parse(trimmed);
893
+ return true;
894
+ } catch (_) {
895
+ return false;
896
+ }
897
+ }
898
+
690
899
  function cleanAIPromptPrefix(text) {
691
900
  if (!text) return "";
692
- let cleaned = text.replace(/^\s*(?:Claude responded|Claude|ChatGPT|Gemini|Grok|DeepSeek|Kimi|Copilot|Assistant|User)\s*(?::|\bsaid\b|\bresponded\b|(?=\s))\s*/i, "");
901
+ let cleaned = text;
902
+ // If text is a full valid JSON object or array, preserve structure
903
+ if (!isJsonPayload(cleaned)) {
904
+ // 1. Strip raw CSS / style blocks leaked from ChatGPT Canvas, web components or stylesheets (handles multi-line, unclosed and variable definitions)
905
+ cleaned = cleaned.replace(/^\s*(?:[.#][a-zA-Z0-9_-]+|\[[a-zA-Z0-9_#.:\-*>=,'"\s]+\]|:is\([^)]+\)|[a-zA-Z0-9_-]+)?\s*\{[^}]*?(?:\}\s*|\n\n+|$)/gi, "");
906
+ cleaned = cleaned.replace(/^[;{} \t\r\n]+/, "");
907
+ cleaned = cleaned.replace(/(?:^|\n)[a-zA-Z0-9_#.:\-*>[\]=\s,'"]+\{[^}]*(--[a-zA-Z0-9_-]+:|color-mix\(|var\()[^}]*\}/g, "");
908
+ }
909
+ // 2. Strip AI author prefixes and platform artifacts
910
+ cleaned = cleaned.replace(/^\s*(?:Claude responded|Claude|ChatGPT|Gemini|Grok|DeepSeek|Kimi|Copilot|Assistant|User)\s*(?::|\bsaid\b|\bresponded\b|(?=\s))\s*/i, "");
693
911
  cleaned = cleaned.replace(/^(?:Here (?:is|are) (?:the )?(?:redacted|scrubbed|sanitized|processed|clean|updated|modified) (?:text|output|version|data).*?[:\n]+|\*\*Scrubbed Text\*\*[:\n]+|### Scrubbed Text[:\n]+)/i, '');
694
912
  cleaned = cleaned.replace(/^\s*Edit\s*\n+/i, "");
695
913
  cleaned = cleaned.replace(/\s*\bEdit\s+in\s+a\s+page\b\s*$/i, "");
914
+ // 3. Strip stray leading colons, semicolons, or separators left by stripped icons/artifact headers
915
+ cleaned = cleaned.replace(/^[:;|\-\—\–]+(?=\n|$)/, "");
916
+ cleaned = cleaned.replace(/^[:;]+\s*/, "");
696
917
  return cleaned.trim();
697
918
  }
698
919
 
920
+ function buildFastTokenLookup(sessionMap) {
921
+ const lookup = new Map();
922
+ const customRegexParts = [];
923
+ const keys = Object.keys(sessionMap || {});
924
+
925
+ for (let i = 0; i < keys.length; i++) {
926
+ const k = keys[i];
927
+ const v = sessionMap[k];
928
+ lookup.set(k, v);
929
+ lookup.set(k.toUpperCase(), v);
930
+
931
+ const inner = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
932
+ const match = inner.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
933
+ if (match) {
934
+ const label = match[1];
935
+ const baseIndex = parseInt(match[2], 10);
936
+ const aliases = getLabelAliases(label);
937
+ for (let a = 0; a < aliases.length; a++) {
938
+ const u = aliases[a].toUpperCase();
939
+ lookup.set(u + '_' + baseIndex, v);
940
+ lookup.set(u + '-' + baseIndex, v);
941
+ lookup.set(u + ' ' + baseIndex, v);
942
+ lookup.set(u + baseIndex, v);
943
+ lookup.set('[' + u + '_' + baseIndex + ']', v);
944
+ lookup.set('<' + u + '_' + baseIndex + '>', v);
945
+ lookup.set('{{' + u + '_' + baseIndex + '}}', v);
946
+ lookup.set('__' + u + '_' + baseIndex + '__', v);
947
+ lookup.set('[' + u + ' ' + baseIndex + ']', v);
948
+ lookup.set('[' + u + '-' + baseIndex + ']', v);
949
+ }
950
+ } else {
951
+ customRegexParts.push(k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
952
+ }
953
+ }
954
+
955
+ let regexStr = '(?:\\[\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*\\]|<\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*>|\\{\\{\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*\\}\\}|__\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*__|(?<=^|[^a-zA-Z0-9_А-Яа-яЁё])[A-Za-z_А-Яа-яЁё]+[-_\\s]*\\d+)';
956
+ if (customRegexParts.length > 0) {
957
+ regexStr = '(?:' + regexStr + '|' + customRegexParts.join('|') + ')';
958
+ }
959
+ const tokenRegex = new RegExp(regexStr + '(?:\'s|’s|s|[а-яёА-ЯЁ]{1,3})?', 'gi');
960
+
961
+ return { lookup, tokenRegex };
962
+ }
963
+
964
+ function resolveTokenValue(rawMatch, targetTokenKey, sessionMap) {
965
+ if (!sessionMap) return undefined;
966
+ if (sessionMap[targetTokenKey] !== undefined) return sessionMap[targetTokenKey];
967
+ if (sessionMap[rawMatch] !== undefined) return sessionMap[rawMatch];
968
+ const cleanRaw = rawMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
969
+ for (const k of Object.keys(sessionMap)) {
970
+ const cleanK = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
971
+ if (cleanK.toLowerCase() === cleanRaw.toLowerCase()) {
972
+ return sessionMap[k];
973
+ }
974
+ }
975
+ return undefined;
976
+ }
977
+
699
978
  function unscrubText(text, sessionMap) {
700
979
  let restoredCount = 0;
701
980
  let result = text;
702
- const tokens = Object.keys(sessionMap);
981
+ const tokens = Object.keys(sessionMap || {});
703
982
  if (tokens.length === 0) return { text: result, count: 0 };
704
983
 
705
984
  result = cleanAIPromptPrefix(result);
706
985
 
986
+ if (tokens.length > 50) {
987
+ const { lookup, tokenRegex } = buildFastTokenLookup(sessionMap);
988
+ result = result.replace(tokenRegex, (match) => {
989
+ if (lookup.has(match)) {
990
+ restoredCount++;
991
+ return lookup.get(match);
992
+ }
993
+ const upper = match.toUpperCase();
994
+ if (lookup.has(upper)) {
995
+ restoredCount++;
996
+ return lookup.get(upper);
997
+ }
998
+ const possMatch = match.match(/^([\s\S]+?)('s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
999
+ if (possMatch) {
1000
+ const base = possMatch[1];
1001
+ const suffix = possMatch[2];
1002
+ if (lookup.has(base)) {
1003
+ restoredCount++;
1004
+ return lookup.get(base) + suffix;
1005
+ }
1006
+ if (lookup.has(base.toUpperCase())) {
1007
+ restoredCount++;
1008
+ return lookup.get(base.toUpperCase()) + suffix;
1009
+ }
1010
+ }
1011
+ return match;
1012
+ });
1013
+ return { text: result, count: restoredCount };
1014
+ }
1015
+
707
1016
  const { compositeRegex, looseRules } = buildRestorationRegexAndRules(sessionMap);
708
1017
 
709
1018
  if (compositeRegex) {
710
1019
  result = result.replace(compositeRegex, (match) => {
711
- const cleanMatch = match.replace(/^\[|\]$/g, '').trim();
712
- const token = `[${cleanMatch.replace(/^\[|\]$/g, '')}]`;
713
- if (sessionMap[token]) {
714
- restoredCount++;
715
- return sessionMap[token];
716
- }
717
- if (sessionMap[cleanMatch]) {
1020
+ const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1021
+ const suffix = suffixMatch ? suffixMatch[0] : '';
1022
+ const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
1023
+ const cleanMatch = baseMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
1024
+ const token = `[${cleanMatch}]`;
1025
+
1026
+ const val = resolveTokenValue(baseMatch, token, sessionMap) ?? resolveTokenValue(baseMatch, cleanMatch, sessionMap);
1027
+ if (val !== undefined) {
718
1028
  restoredCount++;
719
- return sessionMap[cleanMatch];
1029
+ return val + suffix;
720
1030
  }
721
1031
  return match;
722
1032
  });
@@ -725,8 +1035,16 @@ const PROFILE_JARGON = {
725
1035
  looseRules.forEach(rule => {
726
1036
  const rx = new RegExp(rule.pattern, 'gi');
727
1037
  result = result.replace(rx, (match) => {
728
- restoredCount++;
729
- return sessionMap[rule.token];
1038
+ const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1039
+ const suffix = suffixMatch ? suffixMatch[0] : '';
1040
+ const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
1041
+
1042
+ const val = sessionMap[rule.token] ?? resolveTokenValue(baseMatch, rule.token, sessionMap);
1043
+ if (val !== undefined) {
1044
+ restoredCount++;
1045
+ return val + suffix;
1046
+ }
1047
+ return match;
730
1048
  });
731
1049
  });
732
1050
 
@@ -750,21 +1068,50 @@ const PROFILE_JARGON = {
750
1068
  const tokens = Object.keys(sessionMap || {});
751
1069
  if (tokens.length === 0) return { text: result, count: 0 };
752
1070
 
1071
+ if (tokens.length > 50) {
1072
+ const { lookup, tokenRegex } = buildFastTokenLookup(sessionMap);
1073
+ result = result.replace(tokenRegex, (match) => {
1074
+ let rawVal = null;
1075
+ let suffix = '';
1076
+ if (lookup.has(match)) {
1077
+ rawVal = lookup.get(match);
1078
+ } else if (lookup.has(match.toUpperCase())) {
1079
+ rawVal = lookup.get(match.toUpperCase());
1080
+ } else {
1081
+ const possMatch = match.match(/^([\s\S]+?)('s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1082
+ if (possMatch) {
1083
+ const base = possMatch[1];
1084
+ suffix = possMatch[2];
1085
+ if (lookup.has(base)) rawVal = lookup.get(base);
1086
+ else if (lookup.has(base.toUpperCase())) rawVal = lookup.get(base.toUpperCase());
1087
+ }
1088
+ }
1089
+ if (rawVal !== null) {
1090
+ restoredCount++;
1091
+ const escapedVal = escapeHTML(rawVal);
1092
+ const cleanMatch = match.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
1093
+ return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(cleanMatch)}">${escapedVal}</span>${escapeHTML(suffix)}`;
1094
+ }
1095
+ return match;
1096
+ });
1097
+ return { text: result, count: restoredCount };
1098
+ }
1099
+
753
1100
  const { compositeRegex, looseRules } = buildRestorationRegexAndRules(sessionMap);
754
1101
 
755
1102
  if (compositeRegex) {
756
1103
  result = result.replace(compositeRegex, (match) => {
757
- const cleanMatch = match.replace(/^\[|\]$/g, '').trim();
758
- const token = `[${cleanMatch.replace(/^\[|\]$/g, '')}]`;
759
- if (sessionMap[token]) {
760
- restoredCount++;
761
- const escapedVal = escapeHTML(sessionMap[token]);
762
- return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(token)}">${escapedVal}</span>`;
763
- }
764
- if (sessionMap[cleanMatch]) {
1104
+ const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1105
+ const suffix = suffixMatch ? suffixMatch[0] : '';
1106
+ const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
1107
+ const cleanMatch = baseMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
1108
+ const token = `[${cleanMatch}]`;
1109
+
1110
+ const val = resolveTokenValue(baseMatch, token, sessionMap) ?? resolveTokenValue(baseMatch, cleanMatch, sessionMap);
1111
+ if (val !== undefined) {
765
1112
  restoredCount++;
766
- const escapedVal = escapeHTML(sessionMap[cleanMatch]);
767
- return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(cleanMatch)}">${escapedVal}</span>`;
1113
+ const escapedVal = escapeHTML(val);
1114
+ return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(token)}">${escapedVal}</span>${escapeHTML(suffix)}`;
768
1115
  }
769
1116
  return match;
770
1117
  });
@@ -778,9 +1125,17 @@ const PROFILE_JARGON = {
778
1125
  const lastClose = before.lastIndexOf('>');
779
1126
  if (lastOpen > lastClose) return match;
780
1127
 
781
- restoredCount++;
782
- const escapedVal = escapeHTML(sessionMap[rule.token]);
783
- return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(rule.token)} (Fuzzy Match)">${escapedVal}</span>`;
1128
+ const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1129
+ const suffix = suffixMatch ? suffixMatch[0] : '';
1130
+ const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
1131
+
1132
+ const val = sessionMap[rule.token] ?? resolveTokenValue(baseMatch, rule.token, sessionMap);
1133
+ if (val !== undefined) {
1134
+ restoredCount++;
1135
+ const escapedVal = escapeHTML(val);
1136
+ return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(rule.token)} (Fuzzy Match)">${escapedVal}</span>${escapeHTML(suffix)}`;
1137
+ }
1138
+ return match;
784
1139
  });
785
1140
  });
786
1141
 
@@ -798,6 +1153,7 @@ const PROFILE_JARGON = {
798
1153
  LABEL_ALIASES,
799
1154
  getLabelAliases,
800
1155
  PROFILE_ALIAS_MAP,
1156
+ formatToken,
801
1157
  buildRestorationRegexAndRules,
802
1158
  unscrubText,
803
1159
  unscrubTextAsHTML,