@privacyscrubber/mcp-server 1.7.6 → 2.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/ps-pii-engine.cjs CHANGED
@@ -17,39 +17,45 @@ let DEVOPS_SECRETS = [
17
17
  // Secrets & API Keys
18
18
  { name: 'AWS Credentials', type: 'SECRET', regex: /\b(?:AKIA|ASIA|AGPA|AIDA|AROA|AIPA)[A-Z0-9]{16}\b/g },
19
19
  { name: 'JSON Web Token (JWT)', type: 'SECRET', regex: /\beyJ[a-zA-Z0-9_-]+\.[a-zA-Z0-9_-]+\.[a-zA-Z0-9_-]+\b/g },
20
- { name: 'API Token/Key (GitHub/Slack/NPM)', type: 'SECRET', regex: /\b(?:ghp|gho|ghu|ghs|ghr|glpat|npm|xox[baprs])[-_][A-Za-z0-9_]{10,}\b/g },
21
- { name: 'Stripe API Key', type: 'SECRET', regex: /\b(?:[rs]k)_(?:test|live)_[a-zA-Z0-9]{24,}\b/g },
22
- { name: 'Generic Secret/Key', type: 'SECRET', regex: /\b(sk|pk|secret|key|token|auth)(?:[-_][a-zA-Z0-9_-]{5,}|(?=[a-zA-Z0-9_-]{5,}\b)(?=[a-zA-Z_-]*[0-9])[a-zA-Z0-9_-]{5,})\b/gi },
20
+ { name: 'API Token/Key (GitHub/Slack/NPM)', type: 'SECRET', regex: /\b(?:ghp|gho|ghu|ghs|ghr|glpat|npm|xox[baprs])[-_][A-Za-z0-9_-]{10,}\b/g },
21
+ { name: 'Stripe API Key', type: 'SECRET', regex: /\b(?:[rs]k)_(?:test|live)_[a-zA-Z0-9]{14,}\b/g },
22
+ { name: 'OpenAI Project API Key', type: 'SECRET', regex: /\b(?:sk|pk)-(?:proj-)?[a-zA-Z0-9_-]{16,}\b/gi },
23
+ { name: 'Database Connection URI', type: 'SECRET', regex: /\b(?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp|mssql):\/\/[^\s"']+/gi },
24
+ { name: 'Generic Secret/Key', type: 'SECRET', regex: /\b(?:sk|pk|secret|key|token|auth)(?:[-_][a-zA-Z0-9_-]{3,}|(?=[a-zA-Z0-9_-]{5,}\b)(?=[a-zA-Z_-]*[0-9])[a-zA-Z0-9_-]{5,})\b/gi },
23
25
  { name: 'Hash / Hex Key (32-64 chars)', type: 'SECRET', regex: /\b[a-fA-F0-9]{32,64}\b/g },
24
26
  { name: 'CVE Identifier', type: 'SECRET', regex: /\bCVE-\d{4}-\d{4,}\b/gi },
25
- { name: 'Cryptographic Hash', type: 'SECRET', regex: /\b(MD5|SHA1|SHA256)[:\s][a-f0-9]{32,64}\b/gi },
26
- { name: 'Database/API Secret', type: 'SECRET', regex: /\b(DB|POSTGRES|REDIS|MYSQL|AWS|SECRET|PASSWORD|TOKEN|API|KEY)[A-Z0-9_]*\s*[:=]\s*[^\s"']+\b/gi },
27
- { name: 'Proprietary IP / Confidential', type: 'SECRET', regex: /\b(CONFIDENTIAL|PROPRIETARY|TRADE SECRET|DO NOT DISTRIBUTE|INTERNAL USE ONLY)\b/gi },
27
+ { name: 'Cryptographic Hash', type: 'SECRET', regex: /\b(?:MD5|SHA1|SHA256)[:\s][a-f0-9]{32,64}\b/gi },
28
+ { name: 'Database/API Secret', type: 'SECRET', regex: /\b(?:DB|POSTGRES|REDIS|MYSQL|AWS|SECRET|PASSWORD|TOKEN|API|KEY)[A-Z0-9_]*\s*[:=]\s*[^\s"']+\b/gi },
29
+ { name: 'Proprietary IP / Confidential', type: 'SECRET', regex: /\b(?:CONFIDENTIAL|PROPRIETARY|TRADE SECRET|DO NOT DISTRIBUTE|INTERNAL USE ONLY)\b/gi },
28
30
  { name: 'Private Cryptographic Key', type: 'SECRET', regex: /-----BEGIN (?:RSA |EC |PGP |DSA )?PRIVATE KEY-----/g }
29
31
  ];
30
32
 
31
33
  let REGEX_RULES = [
32
34
  ...DEVOPS_SECRETS,
33
35
  // Emails
34
- { type: 'EMAIL', regex: /[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}/g },
36
+ { type: 'EMAIL', regex: /\b[a-zA-Z0-9._%+-]{1,64}@[a-zA-Z0-9.-]{1,255}\.[a-zA-Z]{2,}\b/g },
35
37
 
36
- // Financial Data
37
- { type: 'FINANCIAL', regex: /(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)[0-9,.]+\b/g },
38
- { type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9]([A-Z0-9]{3})?\b/g },
39
- { type: 'FINANCIAL', regex: /\b\d{9}\b/g },
38
+ // Financial Data (PCI-DSS, Bank Accounts, Direct Deposits, Cards, IBAN, SWIFT, Routing Numbers)
39
+ { type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9](?:[A-Z0-9]{3})?\b/g },
40
40
  { type: 'FINANCIAL', regex: /\b[A-Z]{2}[0-9]{2}[a-zA-Z0-9]{4}[0-9]{7}[a-zA-Z0-9]{0,16}\b/g },
41
41
  { type: 'FINANCIAL', regex: /\bPORTFOLIO[-_][A-Z0-9]{5,}\b/gi },
42
42
  { type: 'FINANCIAL', regex: /\b(?:\d[ -]?){13,19}\b/g },
43
43
  { type: 'FINANCIAL', regex: /\b(?:1|3|bc1)[a-zA-HJ-NP-Z0-9]{25,39}\b/g },
44
44
  { type: 'FINANCIAL', regex: /\b0x[a-fA-F0-9]{40}\b/g },
45
+ // Masked / Direct Deposit Bank Account Numbers & Routing Numbers
46
+ { type: 'FINANCIAL', regex: /\b(?:Account|Acct|Checking|Savings|Direct\s+Deposit)\s*(?:#|ID|No\.?|Number)?[:\s#]*(?:[\*xX•.-]{3,}\d{2,6}|\d{4}[-\s]?\d{4}[-\s]?\d{2,6})\b/gi },
47
+ { type: 'FINANCIAL', regex: /\b(?:ABA|Routing|RTN)\s*(?:#|ID|No\.?|Number)?[:\s#]*\d{9}\b/gi },
45
48
 
46
49
  // Legal & Court
47
- { type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9]{4,}\b/gi },
48
- { type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9]{4,}\b/gi },
49
- { type: 'LEGAL', regex: /\b[A-Z]{2,4}[- ]?\d{2}[- ]?\d{4,}\b/g },
50
+ { type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9_-]{4,}\b/gi },
51
+ { type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9_-]{4,}\b/gi },
50
52
  { type: 'PRIVILEGE', regex: /ATTORNEY[- ]CLIENT[- ]PRIVILEGE/gi },
51
53
 
52
- // Professional IDs
54
+ // Professional IDs & Organizations
55
+ { type: 'NAME', isContextName: true, regex: /\b(?:[A-Z][A-Za-z0-9&.,'-]*[ \t\xA0]+){1,5}(?:Inc\.?|LLC|Corp\.?|Corporation|Ltd\.?|Limited|Co\.?|Company|Group|Holdings|Solutions|Services|Technologies|Logistics|Industries|Capital|Bank|Partners|LLP|PLLC)(?:\s+(?:LLC|Inc\.?|Corp\.?|Ltd\.?|USA|Group))?\b/g },
56
+ { type: 'ID', regex: /\b(?:Employee|Emp|EE|Worker|Staff|File|Badge|Member|Advisor|Producer|Agent|Borrower)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Z0-9-]{3,15}\b/gi },
57
+ { type: 'ID', regex: /\b(?:Pay\s+Group|Cost\s+Center|Dept|Department)[:\s#]*[A-Za-z0-9_-]{2,30}\b/gi },
58
+ { type: 'ID', regex: /(?:\b(?:Box\s+d\b|d\.\s*(?:Control|#)?|d\s+Control)\s*(?:number|no\.?|#|num)?[:\s#]*|\bControl\s*(?:number|no\.?|#|num)[:\s#]*|\bControl[:#]\s*)([A-Za-z0-9-]{3,30})/gi },
53
59
  { type: 'ID', regex: /\bEEID[ -]?\d{4,}\b/gi },
54
60
  { type: 'ID', regex: /\bRESUME[-_]?[A-Z0-9]{4,}\b/gi },
55
61
  { type: 'ID', regex: /\bLEAD[-_][A-Z0-9]{5,}\b/gi },
@@ -69,34 +75,52 @@ let REGEX_RULES = [
69
75
  { type: 'ID', regex: /\bCOURSE[-_][A-Z0-9]{4,}\b/gi },
70
76
  { type: 'ID', regex: /\bINSTANCE[-_]ID[-_][a-z0-9-]{10,}\b/gi },
71
77
  { type: 'ID', regex: /\bENV[-_][A-Z0-9]{3,}\b/gi },
72
- { type: 'PRIVACY', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
78
+ { type: 'ID', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
73
79
 
74
- // Addresses & Locations
75
- { type: 'ADDRESS', regex: /\b\d{1,6}\s+(?:[A-Z][a-zA-Z0-9.-]*\s+){1,3}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square)\b/gi },
76
- { type: 'LOCATION', regex: /\b[A-Z][a-zA-Z\s.-]{2,25},\s*[A-Z]{2}\b/g },
80
+ // Insurance & Health Plan IDs
81
+ { type: 'ID', regex: /\b(?:BCB|BCBS|AETNA|CIGNA|UHC|HUMANA|MEDICARE|MEDICAID)[-_A-Za-z0-9]+\b/gi },
82
+ { type: 'ID', regex: /\b(?:Insurance\s+(?:ID|No\.?|Number|#)|Policy(?:\s*(?:ID|No\.?|Number|#)|[:#])|Member\s*(?:ID|No\.?|Number|#|[:#])|Subscriber\s*(?:ID|No\.?|Number|#|[:#])|Group\s*(?:ID|No\.?|Number|#|[:#])|Plan\s*(?:ID|No\.?|Number|#|[:#])|Health(?:\s+Plan)?\s*(?:ID|No\.?|Number|#)|Rx\s*(?:ID|No\.?|Number|Group|BIN|PCN|#))[:\s#]*([A-Za-z0-9-]+)/gi },
83
+ { type: 'ID', regex: /\b(?:Health\s+Plan(?:\s+Beneficiary)?|Beneficiary(?:\s+(?:No\.?|Number|ID|Num|#))?|HPN)[:\s#]+([A-Za-z0-9-]+)/gi },
84
+ { type: 'ID', regex: /\bHPN[-_][A-Za-z0-9-]+\b/gi },
77
85
 
86
+ // Addresses & Locations
87
+ { type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:\bBox\s+f\b|\bf\.\s*|\bf\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?(?:address[,\s]+and\s+ZIP\s+code|address)?|(?:Employee(?:'s)?\s+address[,\s]+and\s+ZIP\s+code))[\s:#]*([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|Box|\d+\b|1\b|2\b|Wages|Federal|Social|Medicare)/gi },
88
+ { type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:Borrower(?:'s)?|Co-Borrower(?:'s)?|Employee(?:'s)?|Employer(?:'s)?|Home|Mailing|Property|Physical)\s+address)[\s:#]+([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|City|State|ZIP|SSN|EIN|Phone|Box|\d+\b)/gi },
89
+ { type: 'ADDRESS', regex: /\b\d{1,6}[ \t\xA0]+(?:[A-Za-z0-9.-]+[ \t\xA0]+){1,4}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Hwy|Highway|Cir|Circle|Trl|Trail|Loop|Row|Pike|Box|PO Box|P\.O\.[ \t\xA0]*Box)\b(?:[ \t\xA0]*,?[ \t\xA0]*(?:Apt|Apartment|Suite|Ste|Unit|#|Fl|Floor|Bldg|Building)\.?[ \t\xA0]*[A-Za-z0-9-]+)?/gi },
90
+ { type: 'ADDRESS', regex: /\b(?:P\.?O\.?[ \t\xA0]*Box|PO[ \t\xA0]*Box)[ \t\xA0]+\d{1,6}\b/gi },
91
+ { type: 'ADDRESS', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
92
+ { type: 'ADDRESS', regex: /\b(?:ZIP|Postal|Code)?\s*(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
93
+ { type: 'ADDRESS', regex: /\b\d{5}-\d{4}\b/g },
94
+ { type: 'LOCATION', regex: /\b[A-Za-z][a-zA-Z .'-]{1,25}(?:,\s*(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)|\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR))\b/g },
78
95
 
79
96
  // PHI & Medical
80
- { type: 'PHI', regex: /\bMRN[ -]?\d{6,}\b/gi },
97
+ { type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record (?:No\.?|Num(?:ber)?|#)|Patient (?:No\.?|Num(?:ber)?|#))[\s:#]+([A-Za-z0-9-]+)/gi },
98
+ { type: 'PHI', regex: /\bMRN[-_ ]*[A-Za-z0-9-]{4,}\b/gi },
99
+ { type: 'ID', regex: /\b(?:NPI|National Provider Identifier)[:\s#]*(\d{10})\b/gi },
100
+ { type: 'ID', regex: /\b(?:Device\s+(?:Identifier|ID|Serial|No\.?|Number)|UDI)[:\s#]+([A-Za-z0-9-]+)/gi },
101
+ { type: 'ID', regex: /\bUDI[-_][A-Za-z0-9-]+\b/gi },
102
+ { type: 'ID', regex: /\b(?:Vehicle\s+(?:Serial|ID|Identification(?:\s+Number)?|No\.?|Number)|VIN)[:\s#]+([A-Za-z0-9-]+)/gi },
103
+ { type: 'ID', regex: /\bVIN[-_][A-Za-z0-9-]+\b/gi },
81
104
  { type: 'PHI', regex: /\b[A-TV-Z]\d{2}[. ]?\d[A-Z0-9]?\b/g },
82
105
  { type: 'PHI', regex: /\b[A-Z]{2,3}\d{6,8}\b/g },
83
106
  { type: 'PHI', regex: /\bNHS[ -]?\d{3}[ -]?\d{3}[ -]?\d{4}\b/gi },
84
107
 
85
-
86
108
  // Copyrights
87
109
  { type: 'COPYRIGHT', regex: /\bPROJECT[-_][A-Z0-9]{5,}\b/gi },
88
110
  { type: 'COPYRIGHT', regex: /\b(DRAFT|ASSET|SCRIPT)[-_][0-9]{4,}\b/gi },
89
111
 
90
- // General Privacy & Dates
91
- { type: 'PRIVACY', regex: /\b(GDPR|HIPAA|CCPA|SOC2)[-_]AUDIT[-_]\d{4}\b/gi },
92
- { type: 'PRIVACY', regex: /\bPOLICY[-_][A-Z0-9]{5,}\b/gi },
93
- { type: 'PRIVACY', regex: /\bGRADE[S]?\s*:\s*[A-DF][+-]?\b/gi },
94
- { type: 'PRIVACY', regex: /\b(DOB|BIRTHDAY)[:\s]*[0-9./-]{6,10}\b/gi },
95
- { type: 'PRIVACY', regex: /\b(PASSWORD|PWD|SECRET)[:\s]*[\S]{4,}\b/gi },
112
+ // General Privacy, Dates & Secrets
113
+ { type: 'ID', regex: /\b(?:GDPR|HIPAA|CCPA|SOC2)[-_]AUDIT[-_]\d{4}\b/gi },
114
+ { type: 'ID', regex: /\bPOLICY[-_][A-Z0-9]{5,}\b/gi },
115
+ { type: 'ID', regex: /\bGRADE[S]?\s*:\s*[A-DF][+-]?\b/gi },
116
+ { type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
117
+ { type: 'SECRET', regex: /\b(?:PASSWORD|PWD|SECRET)\s*[:=]\s*["']?[\S]{4,}["']?/gi },
96
118
 
97
119
  // Standard IDs (SSN, EIN, Passport, VAT)
98
120
  { type: 'ID', regex: /\b\d{3}-\d{2}-\d{4}\b/g },
99
- { type: 'ID', regex: /(?:\/|[A-Za-z]:\\)[\w\-. ]+(?:[\/\\][\w\-. ]+)+/g },
121
+ { type: 'ID', regex: /\b(?:XXX|xxx|\*\*\*)[ -]?(?:XX|xx|\*\*)[ -]?\d{4}\b/g },
122
+ { type: 'ID', regex: /\b\d{2}-\d{7}\b/g },
123
+ { type: 'ID', regex: /(?:(?:[A-Za-z]:\\|\/(?:usr|var|etc|home|root|Users|private|tmp|opt|bin|sbin|dev|Applications|Library)\/)[a-zA-Z0-9_.-]+(?:[\/\\][a-zA-Z0-9_.-]+)*|\/(?:[a-zA-Z0-9_.-]+\/)+[a-zA-Z0-9_.-]+\.(?:txt|pdf|docx|xlsx|csv|js|ts|json|env|log|key|pem|crt|conf|yaml|yml|xml|html|sql|py|go|rs|c|cpp|h|sh|bin|zip|tar|gz|png|jpg|jpeg|svg|webp|wasm)\b)/g },
100
124
  { type: 'ID', regex: /\b[A-CEGHJ-PR-TW-Z]{1}[A-CEGHJ-NPR-TW-Z]{1}[0-9]{6}[A-DFM]{1}\b/gi },
101
125
  { type: 'ID', regex: /\b[A-Z]{2}[0-9]{6,12}\b/gi },
102
126
  { type: 'ID', regex: /[A-Z0-9<]{30,44}/g },
@@ -105,14 +129,13 @@ let REGEX_RULES = [
105
129
  { type: 'IP', regex: /\b(?:\d{1,3}\.){3}\d{1,3}\b/g },
106
130
  { type: 'IP', regex: /\b(?:[a-fA-F0-9]{1,4}:){7}[a-fA-F0-9]{1,4}\b/g },
107
131
  { type: 'ID', regex: /\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b/g },
108
- { type: 'ID', regex: /(?:\B\/|\b[a-zA-Z]:\\)(?:[\w.-]+[\/\\]?)+\b/g },
132
+ { type: 'ID', regex: /(?:\B\/|\b[a-zA-Z]:\\)(?:[\w.-]+[\/\\])*[\w.-]+\b/g },
109
133
  { type: 'ID', regex: /\b\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}(?::\d{2})?)?\b/g },
110
134
  { type: 'ID', regex: /\b\d{4}-\d{2}-\d{2}\b/g },
111
135
  { type: 'ID', regex: /\b\d{2}\/\d{2}\/\d{4}\b/g },
112
136
 
113
- // Phone Numbers — must come after IDs to avoid ID rules stealing phone matches
114
- { type: 'PHONE', regex: /\b(?:\+?\d{1,3}[-.\s]?)?\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}\b/g },
115
- { type: 'PHONE', regex: /(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}/g },
137
+ // Phone Numbers
138
+ { type: 'PHONE', regex: /(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}\b/g },
116
139
  { type: 'PHONE', regex: /\+?[1-9]\d{1,3}[\s.-]\(?\d{1,4}\)?[\s.-]\d{2,4}[\s.-]\d{4}/g },
117
140
  { type: 'PHONE', regex: /(?:\+44\s?7\d{3}|\(?07\d{3}\)?)\s?\d{3}\s?\d{3}\b/g },
118
141
  { type: 'PHONE', regex: /\b(?:\d{3}[-.\s]\d{4}|\(\d{3}\)\s??\d{3}[-.\s]??\d{4}|\d{3}[-.\s]??\d{3}[-.\s]??\d{4})\b/g },
@@ -126,50 +149,70 @@ let REGEX_RULES = [
126
149
  { type: 'ID', regex: /\bDRIVER[S]?\s+LICENSE[ -]?\d{6,15}\b/gi },
127
150
 
128
151
  // Generalized Name Detection (First [Middle] Last) — max 1 middle word to avoid grabbing job titles
129
- // Middle word must be: a particle (van/de/etc.), a single initial (A.), or a capitalized word of ≥2 lowercase letters
130
- { type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]{1,}(?:\p{Lu}[\p{Ll}'-]*)?\p{L}|\p{Lu}\.?|van|von|de|di|da|la|le|del|du|der|van[ \t\xA0]+de)){0,1}[ \t\xA0]+\p{Lu}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
131
- // All-Caps Names (2–3 words, each ≥3 chars) — prevents acronym chains like "GPA ROI AUM" (Unicode-safe)
132
- { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{3,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+\p{Lu}{3,}(?:[\p{Lu}'-]*\p{Lu})?){1,2}(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
133
- // ALL-CAPS first + optional middle initial(s) with optional spaces + ALL-CAPS last name
134
- // E.g. "KKKKK I. MMMMMM", "KKKKK I.MMMMMM", "KKKKKI.MMMMMM" (PDF extraction spacing anomalies)
135
- { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*|[ \t\xA0]+)\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
152
+ // Middle word must be: a particle (van/de/etc.), a single initial (A. or A), or a capitalized word of ≥2 lowercase letters
153
+ { type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*|\p{Lu}\.?|van|von|de|di|da|la|le|del|du|der|van[ \t\xA0]+de|van[ \t\xA0]+der)){0,1}[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
154
+ // Payroll Format Names (e.g. "BARKER, KELLY", "BARKER, KELLY M", "DOE, JOHN M.")
155
+ { type: 'NAME', regex: /\b[A-Z]{2,25},\s+[A-Z]{2,25}(?:\s+[A-Z]\.?|\s+[A-Z]{2,25})*\b/g },
156
+ // All-Caps Names (2–3 words, supporting single-letter middle initial e.g. "KELLY M BARKER", "JOHN M BARKER")
157
+ { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+(?:\p{Lu}\.?|[A-Z][a-z]+))?[ \t\xA0]+\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
158
+ // ALL-CAPS first + middle initial(s) with optional spaces + ALL-CAPS last name
159
+ { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
136
160
  // ALL-CAPS first name + optional middle initial(s) with optional spaces + Mixed-Case last name
137
- // E.g. "JOSEPH I. Casalvieri", "JOSEPH I.Casalvieri", "JOSEPHI.Casalvieri" (PDF extraction spacing anomalies)
138
- { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*|[ \t\xA0]+)\p{Lu}\p{Ll}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
139
- // Single-Word Names with Honorifics (with or without period) (Unicode-safe)
140
- { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])(?:Mr|Mrs|Ms|Dr|Prof|Hon|Mr\.|Mrs\.|Ms\.|Dr\.|Prof\.|Hon\.)[ \t\xA0]+\p{Lu}[\p{Ll}'-]*(?:\p{Lu}[\p{Ll}'-]*)?\p{L}(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
161
+ { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:(?:[ \t\xA0]*\p{Lu}\.)+[ \t\xA0]*|[ \t\xA0]+)(?:\p{Lu}\p{Ll}[\p{Ll}'-]*|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
162
+ // Names with Honorifics (with or without period, supporting single or multi-word full names) (Unicode-safe)
163
+ { type: 'NAME', regex: /(?<=^|[^\p{L}\p{N}_])(?:Mr|Mrs|Ms|Dr|Prof|Hon|Mr\.|Mrs\.|Ms\.|Dr\.|Prof\.|Hon\.)[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*)(?:[ \t\xA0]+(?:\p{Lu}[\p{Ll}'-]*\p{Ll}|\p{Lu}[\p{Ll}'-]*[\p{Lu}'-][\p{Ll}'-]*))?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
141
164
 
142
- // Contextual Names (Emergency Contacts, Spouses, Children, Tenants, etc.) - Uses capture group m[1]
143
- { type: 'NAME', isContextName: true, regex: /(?:Emergency Contact|Contact(?: Name)?|Spouse|Child|Parent|Guardian|Relationship|Kin|Tenant|Landlord|Buyer|Seller|Plaintiff|Defendant|Testator)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g }
165
+ // W-2 Box c Employer Block (Name, Address, and Zip Code)
166
+ { type: 'NAME', isContextName: true, regex: /(?:(?:\bBox\s+c\b|\bc\.\s*|\bc\s+(?=Employer))\s*(?:Employer(?:'s)?\s*)?(?:name[,\s]+address[,\s]+and\s+ZIP\s+code|name)?|(?:Employer(?:'s)?\s+name[,\s]+address[,\s]+and\s+ZIP\s+code))[\s:#]*([A-Za-z0-9&., \t\xA0'-]{2,45}?)(?=\r?\n|$|\s{3,}|\t|EIN|FEIN|Box|\d+\b|Wages|Federal|Social|Medicare)/gi },
167
+ // W-2 Box e Employee Name & Initial
168
+ { type: 'NAME', isContextName: true, regex: /(?:(?:\bBox\s+e\b|\be\.\s*|\be\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?(?:first\s+name(?:\s+(?:and|&)\s+initial)?|name)?[:\s#]*|(?:Employee(?:'s)?\s+first\s+name(?:\s+(?:and|&)\s+initial)?))[\s:#]*([A-Za-z0-9.\s'-]{2,35}?)(?=\r?\n|$|\s{3,}|\t|Last|Surname|Suff|Box|\d+\b|1\b)/gi },
169
+ // Contextual First Names (Employee's first name, First name, Given name)
170
+ { type: 'NAME', isContextName: true, regex: /(?:(?:Employee(?:'s)?|Borrower(?:'s)?|Co-Borrower(?:'s)?|Applicant(?:'s)?|Candidate(?:'s)?|Worker(?:'s)?|Taxpayer(?:'s)?|Spouse(?:'s)?|Person(?:'s)?)\s+)?(?:First\s+name(?:\s+(?:and|&)\s+initial)?|Given\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9.\s'-]{2,30}?)(?=\r?\n|$|\s{3,}|\t|Last|Surname|Family|Suff|Box|Address|SSN|EIN)/gi },
171
+ // Contextual Last Names (Last name, Surname, Family name)
172
+ { type: 'NAME', isContextName: true, regex: /(?:(?:Employee(?:'s)?|Borrower(?:'s)?|Co-Borrower(?:'s)?|Applicant(?:'s)?|Candidate(?:'s)?|Worker(?:'s)?|Taxpayer(?:'s)?|Spouse(?:'s)?|Person(?:'s)?)\s+)?(?:Last\s+name|Surname|Family\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z'-]{2,30})/gi },
173
+ // Contextual General Names (Employee, Borrower, Co-Borrower, Taxpayer, Spouse, Applicant, Candidate, Worker, Employer, Company, Insured, Patient, Client, etc.)
174
+ { type: 'NAME', isContextName: true, regex: /(?:Employee(?:\s+Name)?|Employer(?:\s+Name)?|Borrower(?:\s+Name)?|Co-Borrower(?:\s+Name)?|Applicant(?:\s+Name)?|Candidate(?:\s+Name)?|Worker(?:\s+Name)?|Taxpayer(?:\s+Name)?|Spouse(?:\s+Name)?|Manager|Supervisor|Reporting To|Insured|Claimant|Patient|Client|Customer|Account Holder|Prepared By|Attention|Attn|Contact(?: Name)?|Child|Parent|Guardian|Relationship|Kin|Tenant|Landlord|Buyer|Seller|Plaintiff|Defendant|Testator)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9&.,\s'-]{2,40}?)(?=\r?\n|$|\s{3,}|\t|Employee|Employer|Address|Phone|SSN|EIN|FEIN|Date|Pay|Rate|Tax|W-2|OMB|Copy|Box|Status)/gi },
175
+ // Box e shorthand
176
+ { type: 'NAME', isContextName: true, regex: /\b(?:Box\s+e)\s*[:#-]\s*([A-Za-z0-9&.,\s'-]{2,40})/gi }
144
177
  ];
145
178
 
146
179
  let PROFILE_RULES = {
147
180
  general: [],
148
181
  legal: [
149
- { type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9]{4,}\b/gi },
150
- { type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9]{4,}\b/gi },
182
+ { type: 'LEGAL', regex: /\bCASE[-_][A-Z0-9_-]{4,}\b/gi },
183
+ { type: 'LEGAL', regex: /\bMATTER[-_][A-Z0-9_-]{4,}\b/gi },
151
184
  { type: 'LEGAL', regex: /\b[A-Z]{2,4}[- ]?\d{2}[- ]?\d{4,}\b/g },
152
185
  { type: 'PRIVILEGE', regex: /ATTORNEY[- ]CLIENT[- ]PRIVILEGE/gi }
153
186
  ],
154
187
  hr: [
155
- { type: 'NAME', isContextName: true, regex: /(?:Candidate|Applicant|Employee|Reporting To|Manager)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g },
156
-
188
+ { type: 'NAME', isContextName: true, regex: /(?:Candidate|Applicant|Employee|Reporting To|Manager|Mentored by|Direct Report)[\s:]+([A-Z][a-z]*(?:\s+[A-Z][a-z]*)?)/g },
157
189
  { type: 'ID', regex: /\bEEID[ -]?\d{4,}\b/gi },
158
190
  { type: 'ID', regex: /\bEMP[-_]\d{3,}\b/gi },
159
191
  { type: 'ID', regex: /\bRESUME[-_]?[A-Z0-9]{4,}\b/gi },
160
- { type: 'PRIVACY', regex: /\\b(DOB|BIRTHDAY)[:\\s]*[0-9./-]{6,10}\\b/gi },
161
- { type: 'ADDRESS', regex: /\\b\\d{1,6}\\s+(?:[A-Z][a-zA-Z0-9.-]*\\s+){1,3}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square)\\b/gi }
192
+ { type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
193
+ { type: 'DATE', regex: /\b(?:Graduated|Graduation|Class of)[:\s]+(?:(?:Spring|Summer|Fall|Winter)\s+)?(?:(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec|January|February|March|April|June|July|August|September|October|November|December)\s+)?\d{4}\b/gi },
194
+ { type: 'ID', regex: /(?:https?:\/\/)?(?:www\.)?linkedin\.com\/in\/[A-Za-z0-9_-]+/gi },
195
+ { type: 'ID', regex: /(?:https?:\/\/)?(?:www\.)?github\.com\/[A-Za-z0-9_-]+/gi },
196
+ { type: 'ADDRESS', regex: /\b\d{1,6}\s+(?:[A-Z0-9][a-zA-Z0-9-]*\s+){1,3}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Highway|Hwy|Circle|Cir|Trail|Trl|Dr(?!\.?\s+[A-Z][a-z]+))\b/g }
162
197
  ],
163
198
  finance: [
164
- { type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9]([A-Z0-9]{3})?\b/g },
199
+ { type: 'FINANCIAL', regex: /\b[A-Z]{4}(?:AD|AE|AF|AG|AI|AL|AM|AO|AQ|AR|AS|AT|AU|AW|AX|AZ|BA|BB|BD|BE|BF|BG|BH|BI|BJ|BL|BM|BN|BO|BQ|BR|BS|BT|BV|BW|BY|BZ|CA|CC|CD|CF|CG|CH|CI|CK|CL|CM|CN|CO|CR|CU|CV|CW|CX|CY|CZ|DE|DJ|DK|DM|DO|DZ|EC|EE|EG|EH|ER|ES|ET|FI|FJ|FK|FM|FO|FR|GA|GB|GD|GE|GF|GG|GH|GI|GL|GM|GN|GP|GQ|GR|GS|GT|GU|GW|GY|HK|HM|HN|HR|HT|HU|ID|IE|IL|IM|IN|IO|IQ|IR|IS|IT|JE|JM|JO|JP|KE|KG|KH|KI|KM|KN|KP|KR|KW|KY|KZ|LA|LB|LC|LI|LK|LR|LS|LT|LU|LV|LY|MA|MC|MD|ME|MF|MG|MH|MK|ML|MM|MN|MO|MP|MQ|MR|MS|MT|MU|MV|MW|MX|MY|MZ|NA|NC|NE|NF|NG|NI|NL|NO|NP|NR|NU|NZ|OM|PA|PE|PF|PG|PH|PK|PL|PM|PN|PR|PS|PT|PW|PY|QA|RE|RO|RS|RU|RW|SA|SB|SC|SD|SE|SG|SH|SI|SJ|SK|SL|SM|SN|SO|SR|SS|ST|SV|SX|SY|SZ|TC|TD|TF|TG|TJ|TK|TL|TM|TN|TO|TR|TT|TV|TW|TZ|UA|UG|UM|US|UY|UZ|VA|VC|VE|VG|VI|VN|VU|WF|WS|YE|YT|ZA|ZM|ZW)[A-Z2-9][A-NP-Z0-9](?:[A-Z0-9]{3})?\b/g },
165
200
  { type: 'FINANCIAL', regex: /\b[A-Z]{2}[0-9]{2}[a-zA-Z0-9]{4}[0-9]{7}[a-zA-Z0-9]{0,16}\b/g },
166
201
  { type: 'FINANCIAL', regex: /\bPORTFOLIO[-_][A-Z0-9]{5,}\b/gi }
167
202
  ],
168
203
  medical: [
169
- { type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record No|Patient No)[\s:#]*([A-Za-z0-9-]+)/gi },
170
- { type: 'PRIVACY', regex: /\b(DOB|Date of Birth)[:\s]*[0-9./-]{6,10}\b/gi },
171
-
172
- { type: 'PHI', regex: /\bMRN[ -]?\d{6,}\b/gi },
204
+ { type: 'PHI', regex: /\b(?:MRN|Patient ID|Medical Record (?:No\.?|Num(?:ber)?|#)|Patient (?:No\.?|Num(?:ber)?|#))[\s:#]+([A-Za-z0-9-]+)/gi },
205
+ { type: 'DATE', regex: /\b(?:DOB|Date of Birth|BIRTHDAY)[\s:]+([0-9./-]{6,10})\b/gi },
206
+ { type: 'PHI', regex: /\bMRN[-_ ]*[A-Za-z0-9-]{4,}\b/gi },
207
+ { type: 'ID', regex: /\b(?:Insurance\s+(?:ID|No\.?|Number|#)|Policy(?:\s*(?:ID|No\.?|Number|#)|[:#])|Member\s*(?:ID|No\.?|Number|#|[:#])|Subscriber\s*(?:ID|No\.?|Number|#|[:#])|Group\s*(?:ID|No\.?|Number|#|[:#])|Plan\s*(?:ID|No\.?|Number|#|[:#])|Health(?:\s+Plan)?\s*(?:ID|No\.?|Number|#)|Rx\s*(?:ID|No\.?|Number|Group|BIN|PCN|#))[:\s#]*([A-Za-z0-9-]+)/gi },
208
+ { type: 'ID', regex: /\b(?:Health\s+Plan(?:\s+Beneficiary)?|Beneficiary(?:\s+No\.?|\s+Number)?|HPN)[:\s#]+([A-Za-z0-9-]+)/gi },
209
+ { type: 'ID', regex: /\bHPN[-_][A-Za-z0-9-]+\b/gi },
210
+ { type: 'ID', regex: /\b(?:BCB|BCBS|AETNA|CIGNA|UHC|HUMANA|MEDICARE|MEDICAID)[-_A-Za-z0-9]+\b/gi },
211
+ { type: 'ID', regex: /\b(?:NPI|National Provider Identifier)[:\s#]*(\d{10})\b/gi },
212
+ { type: 'ID', regex: /\b(?:Device\s+(?:Identifier|ID|Serial|No\.?|Number)|UDI)[:\s#]+([A-Za-z0-9-]+)/gi },
213
+ { type: 'ID', regex: /\bUDI[-_][A-Za-z0-9-]+\b/gi },
214
+ { type: 'ID', regex: /\b(?:Vehicle\s+(?:Serial|ID|Identification(?:\s+Number)?|No\.?|Number)|VIN)[:\s#]+([A-Za-z0-9-]+)/gi },
215
+ { type: 'ID', regex: /\bVIN[-_][A-Za-z0-9-]+\b/gi },
173
216
  { type: 'PHI', regex: /\b[A-TV-Z]\d{2}[. ]?\d[A-Z0-9]?\b/g },
174
217
  { type: 'PHI', regex: /\b[A-Z]{2,3}\d{6,8}\b/g },
175
218
  { type: 'PHI', regex: /\bNHS[ -]?\d{3}[ -]?\d{3}[ -]?\d{4}\b/gi }
@@ -186,13 +229,13 @@ let PROFILE_RULES = {
186
229
  bizops: [
187
230
  { type: 'ID', regex: /\b(?:DEAL|KPI|METRIC)[-_: ]?[A-Z0-9]{4,}\b/gi },
188
231
  { type: 'ID', regex: /\b(?:ENTITY|VENDOR|PARTNER)[-_: ]?[0-9]{4,10}\b/gi },
189
- { type: 'FINANCIAL', regex: /\b(?:REVENUE|EBITDA|PROFIT|MARGIN)[-_: ]?(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
232
+ { type: 'FINANCIAL', regex: /\b(?:REVENUE|EBITDA|PROFIT|MARGIN)[\s:_-]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
190
233
  { type: 'SECRET', regex: /\b(?:NDA|M&A|MERGER)[-_: ]?[A-Z0-9]{4,}\b/gi }
191
234
  ],
192
235
  sales: [
193
236
  { type: 'ID', regex: /\bOPPORTUNITY[-_: ]?[A-Z0-9]{5,}\b/gi },
194
237
  { type: 'ID', regex: /\b(?:DOCUSIGN|CONTRACT)[-_: ]?[0-9A-F]{8,32}\b/gi },
195
- { type: 'FINANCIAL', regex: /\b(?:ARR|MRR|QUOTA)[\s:]?(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
238
+ { type: 'FINANCIAL', regex: /\b(?:ARR|MRR|QUOTA)[\s:]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)?[0-9,.]+[KM]?\b/gi },
196
239
  { type: 'ID', regex: /\b(?:SFDC|HUBSPOT)[-_: ]?[0-9A-Z]{15,18}\b/gi }
197
240
  ],
198
241
  support: [
@@ -205,8 +248,8 @@ let PROFILE_RULES = {
205
248
  { type: 'ID', regex: /\b(?:MLS|LIS)[- ]?\d{6,10}\b/gi },
206
249
  { type: 'ID', regex: /\bPARCEL[- ]?\d{5,15}\b/gi },
207
250
  { type: 'ID', regex: /\bTENANT[-_]ID[-_][0-9]{4,}\b/gi },
208
- { type: 'FINANCIAL', regex: /\b(?:RENT|LEASE|ESCROW)[\s:]?(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)[0-9,]{3,}\b/gi },
209
- { type: 'SECRET', regex: /\b(?:GATE|DOOR|LOBBY)[-_ ](?:CODE|PIN)[\s:]?\d{4,6}\b/gi }
251
+ { type: 'FINANCIAL', regex: /\b(?:RENT|LEASE|ESCROW)[\s:]+(?:[$€£¥₪₽₹]|(?:USD|EUR|GBP|CHF|ILS|RUB)\s?)[0-9,]{3,}\b/gi },
252
+ { type: 'SECRET', regex: /\b(?:GATE|DOOR|LOBBY)[-_ ](?:CODE|PIN)[\s:]*\d{4,6}\b/gi }
210
253
  ],
211
254
  compliance: [
212
255
  { type: 'SECRET', regex: /\b(?:GDPR|HIPAA|CCPA|SOC2|ISO27001)[-_: ]?AUDIT[-_: ]?\d{4}\b/gi },
@@ -214,10 +257,10 @@ let PROFILE_RULES = {
214
257
  { type: 'ID', regex: /\b(?:SAR|DSAR)[-_\/: ]?[A-Z0-9-/]+\b/gi }
215
258
  ],
216
259
  ccpa: [
217
- { type: 'ID', regex: /\bDL[ -]?\d{6,12}\b/gi },
260
+ { type: 'ID', regex: /\b(?:DL|DRIVER['’]?S?\s+LICENSE)[:\s#-]*[A-Z0-9]{6,12}\b/gi },
218
261
  { type: 'LOCATION', regex: /\b-?\d{1,3}\.\d{4,6}[° ]?[NSns],\s*-?\d{1,3}\.\d{4,6}[° ]?[EWew]\b/g },
219
- { type: 'PRIVACY', regex: /\b(?:CCPA|CPRA)[-_: ]?OPT[-_ ]OUT\b/gi },
220
- { type: 'ID', regex: /\bACCOUNT[ -]?(?:ID|NUM|NUMBER)[:\s][A-Z0-9]{6,20}\b/gi }
262
+ { type: 'ID', regex: /\b(?:CCPA|CPRA)[-_: ]?OPT[-_ ]OUT\b/gi },
263
+ { type: 'ID', regex: /\bACCOUNT[ -]?(?:ID|NUM|NUMBER)[:\s]+[A-Z0-9]{6,20}\b/gi }
221
264
  ],
222
265
  engineering: [
223
266
  { type: 'SECRET', regex: /(?<=\b(?:DB|POSTGRES|REDIS|MYSQL|AWS|SECRET|PASSWORD|TOKEN|API|KEY)[A-Z0-9_]*\s*[:=]\s*["']?)[A-Za-z0-9_-]{10,}/gi },
@@ -226,7 +269,7 @@ let PROFILE_RULES = {
226
269
  { type: 'ID', regex: /\b[a-z0-9](?:[-a-z0-9]*[a-z0-9])?\.svc\.cluster\.local\b/g }
227
270
  ],
228
271
  agents: [
229
- { type: 'ID', regex: /\b(?:AGENT|VECTOR|EMBEDDING)[-_: ]?[A-Z0-9]{8,}\b/gi },
272
+ { type: 'ID', regex: /\b(?:AGENT|VECTOR|EMBEDDING)[-_: ]?(?:ID[-_: ]?)?[A-Z0-9]{8,}\b/gi },
230
273
  { type: 'ID', regex: /\bTASK[-_: ]?[A-Z0-9]{5,15}\b/gi },
231
274
  { type: 'SECRET', regex: /\b(?:SYS_PROMPT|SYSTEM_PROMPT|OPENAI_API_KEY)[-_: ]?[A-Za-z0-9_-]{10,}\b/gi }
232
275
  ],
@@ -247,7 +290,7 @@ let PROFILE_RULES = {
247
290
  { type: 'SECRET', regex: /\b(?:CONFIG|KUBECONFIG|TFSTATE)[-_: ]?[A-Z0-9]{6,15}\b/gi }
248
291
  ],
249
292
  personal: [
250
- { type: 'ID', regex: /\b(?:DOB|BIRTHDAY)[\s:]*[0-9./-]{6,10}\b/gi },
293
+ { type: 'DATE', regex: /\b(?:DOB|BIRTHDAY|Date of Birth)[\s:]+([0-9./-]{6,10})\b/gi },
251
294
  { type: 'SECRET', regex: /\b(?:PASSWORD|PWD|SECRET|PIN)[\s:]*[\S]{4,20}\b/gi },
252
295
  { type: 'PHONE', regex: /\b(?:WIFE|HUSBAND|PARTNER|MOM|DAD)[\s:]+(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}\b/gi }
253
296
  ],
@@ -294,10 +337,46 @@ let PROFILE_RULES = {
294
337
  { type: 'ID', regex: /\b(?:Batch|Lot|Serial)\s*(?:No[.]?|Number|#)?[:\s#]*[A-Z0-9][A-Z0-9\-]{3,14}\b/gi },
295
338
  { type: 'PHI', regex: /\b(?:Dose|Dosage)[:\s]+\d+(?:\.\d+)?\s*(?:mg|mcg|mL|IU|units?)\b/gi },
296
339
  { type: 'ID', regex: /\b(?:CRF|eCRF|Case\s+Report\s+Form)\s*(?:No|Page|ID)?[:\s#]*[A-Z0-9]{2,10}\b/gi }
340
+ ],
341
+ underwriting: [
342
+ // Employer Corporate / Business Names
343
+ { type: 'NAME', isContextName: true, regex: /\b(?:[A-Z][A-Za-z0-9&.,'-]*[ \t\xA0]+){1,5}(?:Inc\.?|LLC|Corp\.?|Corporation|Ltd\.?|Limited|Co\.?|Company|Group|Holdings|Solutions|Services|Technologies|Logistics|Industries|Capital|Bank|Partners|LLP|PLLC)(?:\s+(?:LLC|Inc\.?|Corp\.?|Ltd\.?|USA|Group))?\b/g },
344
+ { type: 'NAME', isContextName: true, regex: /(?:Employer|Company|Organization|Business)\s*(?:Name)?[\s:#]+([A-Za-z0-9&.,\s'-]{2,40}?)(?=\r?\n|$|\s{3,}|\t|Address|EIN|FEIN|Phone|W-2|Rate|Pay|Wage)/gi },
345
+
346
+ // W-2 & Tax Identifiers
347
+ { type: 'ID', regex: /(?:\b(?:Box\s+d\b|d\.\s*(?:Control|#)?|d\s+Control)\s*(?:number|no\.?|#|num)?[:\s#]*|\bControl\s*(?:number|no\.?|#|num)[:\s#]*|\bControl[:#]\s*)([A-Za-z0-9-]{3,30})/gi },
348
+ { type: 'ID', regex: /\b(?:EIN|FEIN|Tax\s+ID)[:\s#]*\d{2}-\d{7}\b/gi },
349
+ { type: 'ID', regex: /\b\d{2}-\d{7}\b/g },
350
+ { type: 'ID', regex: /\b\d{3}-\d{2}-\d{4}\b/g },
351
+ { type: 'ID', regex: /\b(?:XXX|xxx|\*\*\*)[ -]?(?:XX|xx|\*\*)[ -]?\d{4}\b/g },
352
+
353
+ // Employee, Loan & Payroll IDs
354
+ { type: 'ID', regex: /\b(?:Employee|Emp|EE|Worker|Borrower|Badge|Advisor|Producer|Agent|Applicant|File)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Z0-9-]{3,20}\b/gi },
355
+ { type: 'ID', regex: /\b(?:Pay\s+Group|Cost\s+Center|Dept|Department)[:\s#]*[A-Za-z0-9_-]{2,30}\b/gi },
356
+ { type: 'ID', regex: /\b(?:Loan|Application|Deal|Borrower|File)\s*(?:#|ID|No\.?|Number)[:\s#]*[A-Za-z0-9-]{4,25}\b/gi },
357
+
358
+ // Direct Deposit, Bank Accounts & Routing Numbers (Masked & Unmasked)
359
+ { type: 'FINANCIAL', regex: /\b(?:Account|Acct|Checking|Savings|Direct\s+Deposit)\s*(?:#|ID|No\.?|Number)?[:\s#]*(?:[\*xX•.-]{3,}\d{2,6}|\d{4}[-\s]?\d{4}[-\s]?\d{2,6})\b/gi },
360
+ { type: 'FINANCIAL', regex: /\b(?:ABA|Routing|RTN)\s*(?:#|ID|No\.?|Number)?[:\s#]*\d{9}\b/gi },
361
+
362
+ // Borrower & Co-Borrower Names (ALL-CAPS, Payroll, Title Case)
363
+ { type: 'NAME', regex: /\b[A-Z]{2,25},\s+[A-Z]{2,25}(?:\s+[A-Z]\.?|\s+[A-Z]{2,25})*\b/g },
364
+ { type: 'NAME', isAggressiveName: true, regex: /(?<=^|[^\p{L}\p{N}_])\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:[ \t\xA0]+(?:\p{Lu}\.?|[A-Z][a-z]+))?[ \t\xA0]+\p{Lu}{2,}(?:[\p{Lu}'-]*\p{Lu})?(?:'s)?(?=[^\p{L}\p{N}_]|$)(?![ \t\xA0]*:)/gu },
365
+ { type: 'NAME', isContextName: true, regex: /(?:(?:Borrower|Co-Borrower|Applicant|Co-Applicant|Employee|Worker|Taxpayer|Candidate|Primary\s+Borrower|Joint\s+Borrower|Account\s+Holder|Insured|Client)\s*(?:Name)?|First\s+name|Given\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z0-9&.,\s'-]{2,35}?)(?=\r?\n|$|\s{3,}|\t|SSN|EIN|DOB|Address|Phone|Rate|Pay|Wage|Date|Box|Last|Surname)/gi },
366
+ { type: 'NAME', isContextName: true, regex: /(?:Last\s+name|Surname|Family\s+name)[\s:#]+(?:\b|\b\s*)([A-Za-z'-]{2,30})/gi },
367
+
368
+ // Addresses & Locations
369
+ { type: 'ADDRESS', isContextAddress: true, regex: /(?:(?:Borrower(?:'s)?|Co-Borrower(?:'s)?|Employee(?:'s)?|Employer(?:'s)?|Home|Mailing|Property|Physical)\s+address|(?:(?:\bBox\s+f\b|\bf\.\s*|\bf\s+(?=Employee))\s*(?:Employee(?:'s)?\s*)?address))[\s:#]+([A-Za-z0-9#.,\s-]{4,55}?)(?=\r?\n|$|\s{3,}|\t|City|State|ZIP|SSN|EIN|Phone|Box|\d+\b)/gi },
370
+ { type: 'ADDRESS', regex: /\b\d{1,6}[ \t\xA0]+(?:[A-Za-z0-9.-]+[ \t\xA0]+){1,4}(?:St|Street|Ave|Avenue|Blvd|Boulevard|Rd|Road|Ln|Lane|Dr|Drive|Way|Ct|Court|Pl|Place|Terrace|Pkwy|Parkway|Sq|Square|Hwy|Highway|Cir|Circle|Trl|Trail|Loop|Row|Pike|Box|PO Box|P\.O\.[ \t\xA0]*Box)\b(?:[ \t\xA0]*,?[ \t\xA0]*(?:Apt|Apartment|Suite|Ste|Unit|#|Fl|Floor|Bldg|Building)\.?[ \t\xA0]*[A-Za-z0-9-]+)?/gi },
371
+ { type: 'ADDRESS', regex: /\b[A-Za-z][a-zA-Z\s.-]{1,25},?\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)\s+\d{5}(?:-\d{4})?\b/g },
372
+ { type: 'ADDRESS', regex: /\b\d{5}-\d{4}\b/g },
373
+ { type: 'LOCATION', regex: /\b[A-Za-z][a-zA-Z .'-]{1,25}(?:,\s*(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|ID|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR)|\s+(?:AL|AK|AZ|AR|CA|CO|CT|DE|FL|GA|HI|IL|IN|IA|KS|KY|LA|ME|MD|MA|MI|MN|MS|MO|MT|NE|NV|NH|NJ|NM|NY|NC|ND|OH|OK|OR|PA|RI|SC|SD|TN|TX|UT|VT|VA|WA|WV|WI|WY|DC|PR))\b/g }
297
374
  ]
298
375
  };
299
376
 
300
377
  let NAME_STOP_LIST = new Set([
378
+ 'clinical note', 'case note', 'prod log', 'siem alert', 'critical security incident', 'security incident', 'hr review', 'crm export', 'bank statement', 'file export', 'database row', 'lease application', 'strategy export', 'action log', 'glossary query', 'tool comparison', 'call transcript', 'zendesk ticket', 'board minutes', 'agent context', 'config dump', 'database dump', 'patient note', 'medical record', 'admission note', 'discharge summary', 'progress note', 'hiring review', 'security audit', 'incident response', 'server log', 'system log', 'api response', 'error log', 'audit log', 'debug log',
379
+ 'tax statement', 'wage and tax statement', 'wage and tax', 'wage statement', 'earning statement', 'earnings statement', 'pay statement', 'pay stub', 'paystub', 'withholding statement',
301
380
  'case no', 'account no', 'client no', 'ref no', 'matter no',
302
381
  'affected user', 'incident date', 'incident type', 'incident report',
303
382
  'review period', 'review date', 'salary band', 'salary range',
@@ -333,13 +412,15 @@ let NAME_STOP_LIST = new Set([
333
412
  'first name', 'last name', 'middle name', 'full name', 'email address', 'phone number', 'cell phone', 'home phone', 'zip code', 'postal code', 'page number', 'section one', 'table contents', 'table of', 'figure one',
334
413
  'marketing department', 'sales department', 'engineering team', 'product team', 'customer support', 'human resources', 'public relations',
335
414
  'artificial intelligence', 'machine learning', 'deep learning', 'large language', 'operating system', 'source code', 'user interface', 'web browser', 'pull request', 'merge request', 'commit message', 'code review', 'cloud computing', 'database schema',
336
- 'blood pressure', 'heart rate', 'chief physician', 'treating physician', 'health care', 'healthcare provider', 'medical record',
415
+ 'blood pressure', 'heart rate', 'chief physician', 'treating physician', 'health care', 'healthcare provider', 'medical record', 'medical record number', 'acute bronchitis', 'vital signs', 'vital sign', 'health plan', 'health plan beneficiary', 'device identifier', 'vehicle serial',
337
416
  'grade a', 'grade b', 'grade c', 'grade d', 'grade f',
338
417
  'version 1', 'version 2', 'version 3', 'version 4', 'version 5',
339
418
  'step 1', 'step 2', 'step 3', 'step 4', 'step 5',
340
419
  'page 1', 'page 2', 'page 3', 'page 4', 'page 5',
341
420
  'cs101', 'course cs101',
342
- // Expanded Stop List (Common nouns that look like names)
421
+ // Expanded Stop List (Common nouns, command phrases, legal, prompt, chess, and animation terms)
422
+ 'docket number', 'docket numbers', 'dockets section', 'case name', 'case names', 'case number', 'case numbers', 'law firm', 'law firms', 'counsel stack', 'counselstack', 'counselstack connector', 'tier 0', 'tier 1', 'tier 2', 'tier 3', 'tier 4', 'do not', 'do not write', 'specific permission', 'write again', 'without permission', 'without specific permission', 'on screen', 'in report', 'own line', 'connector access', 'prompt instruction', 'prompt instructions', 'finding report', 'findings report',
423
+ 'white bishop', 'black bishop', 'white knight', 'black knight', 'white king', 'black king', 'white queen', 'black queen', 'white rook', 'black rook', 'white pawn', 'black pawn', 'chess piece', 'chess pieces', 'chess game', 'chess match', 'disney-pixar', 'disney pixar', 'pixar animation', 'close-up', 'close up',
343
424
  'january', 'february', 'march', 'april', 'may', 'june', 'july', 'august', 'september', 'october', 'november', 'december',
344
425
  'monday', 'tuesday', 'wednesday', 'thursday', 'friday', 'saturday', 'sunday',
345
426
  'yesterday', 'tomorrow', 'today', 'last week', 'next month', 'early morning', 'late night',
@@ -373,24 +454,115 @@ let NAME_STOP_LIST = new Set([
373
454
  'class name', 'function name', 'variable name', 'database table', 'schema name', 'index name', 'query result', 'error message', 'warning message', 'log entry', 'debug log', 'stack trace',
374
455
  'staff member', 'team member', 'board member', 'board meeting', 'committee member', 'executive board',
375
456
  'email us', 'contact us', 'about us', 'sign in', 'sign out',
457
+ 'driver license', 'drivers license', 'opt out', 'opt-out', 'ccpa opt', 'cpra opt', 'spoiler unreleased', 'unreleased draft',
376
458
  'lighting', 'keyboard', 'creating', 'building', 'training', 'planning', 'starting', 'painting', 'printing', 'returned', 'released', 'required', 'accepted', 'imported', 'services', 'products', 'accounts', 'settings', 'partners', 'keywords', 'keystone', 'keyspace', 'keynotes', 'keychain'
377
- , 'quarterly results', 'strategic planning', 'market research', 'customer base', 'privacy settings', 'account settings', 'security settings', 'download now', 'free trial', 'limited time', 'copyright protected', 'all rights', 'rights reserved', 'credit score', 'monthly rent', 'lease application', 'property address', 'reference number', 'additional identifier', 'lease agreement', 'hiring review', 'candidate name', 'privacy policy', 'terms of service', 'machine learning', 'artificial intelligence', 'generative ai', 'silicon valley', 'google cloud', 'amazon web', 'data science', 'operating system', 'software engineer', 'product manager', 'project manager', 'data analyst', 'gross margin', 'revenue growth', 'source code', 'version control', 'large language model']);
459
+ , 'quarterly results', 'strategic planning', 'market research', 'customer base', 'privacy settings', 'account settings', 'security settings', 'download now', 'free trial', 'limited time', 'copyright protected', 'all rights', 'rights reserved', 'credit score', 'monthly rent', 'lease application', 'property address', 'reference number', 'additional identifier', 'lease agreement', 'hiring review', 'candidate name', 'privacy policy', 'terms of service', 'machine learning', 'artificial intelligence', 'generative ai', 'silicon valley', 'google cloud', 'amazon web', 'data science', 'operating system', 'software engineer', 'product manager', 'project manager', 'data analyst', 'gross margin', 'revenue growth', 'source code', 'version control', 'large language model',
460
+ 'wages', 'wage', 'tips', 'compensation', 'withheld', 'withholding', 'medicare', 'deductions', 'deduction',
461
+ 'regular', 'hours', 'holiday', 'overtime', 'commission', 'bonus', 'bonuses', 'records', 'record', 'statement', 'statements', 'rate', 'rates', 'current', 'ytd', 'benefits', 'taxable', 'pre-tax', 'post-tax', 'reimbursements', 'reimbursement', 'fica', 'oasdi', 'disability', 'unemployment', 'sui', 'sdi', 'std', 'ltd', 'exemptions', 'exemption', 'allowances', 'allowance', 'filing', 'status', 'single', 'married', 'head', 'household', 'advice', 'frequency', 'bi-weekly', 'biweekly', 'weekly', 'monthly', 'semi-monthly', 'direct', 'deposit', 'routing', 'box', 'boxes', 'code', 'control', 'omb', 'copy', 'instructions', 'information', 'deferred', 'adoption', 'statutory', 'third-party', 'sick', 'form', 'schedule', 'w-2', 'w2', 'w-4', 'w4', '1099', 'k-1', '1040', 'fed', 'med', 'fwt', 'swt', 'fed w/h', 'fed med', 'locality', 'state wages', 'state tax', 'local wages', 'local tax', 'allocated', 'nonqualified', 'suff', 'suffix', 'allocated tips', 'advance eic', 'advance eic payment', 'dependent care', 'dependent care benefits', 'nonqualified plans', 'statutory employee', 'retirement plan', 'third-party sick pay']);
378
462
 
379
- let JARGON_WORDS = new Set(['step', 'page', 'grade', 'version', 'course', 'class', 'follow', 'chapter', 'lesson', 'unit', 'marketing', 'manager', 'specialist', 'science', 'administration', 'university', 'skills', 'leadership', 'communication', 'working', 'proficiency', 'decision', 'driven', 'experience', 'summary', 'bachelor', 'ads', 'solutions', 'positioning', 'acquisition', 'strategy', 'research', 'database', 'forecast', 'interest', 'prepared', 'merchant', 'document', 'feedback', 'template', 'campaign', 'partners', 'settings', 'keystone', 'llm', 'gpt', 'chatgpt', 'openai', 'anthropic', 'claude', 'gemini', 'api', 'json', 'xml', 'html', 'css', 'javascript', 'python', 'policy', 'terms', 'conditions', 'release', 'sprint', 'deployment', 'cluster', 'instance', 'package', 'module', 'revenue', 'margin', 'gross', 'quarter', 'system', 'code', 'data', 'cloud', 'server', 'database', 'artificial', 'intelligence', 'learning', 'generative']);
463
+ let JARGON_WORDS = new Set(['step', 'page', 'grade', 'version', 'course', 'class', 'follow', 'chapter', 'lesson', 'unit', 'marketing', 'manager', 'specialist', 'science', 'administration', 'university', 'skills', 'leadership', 'communication', 'working', 'proficiency', 'decision', 'driven', 'experience', 'summary', 'bachelor', 'ads', 'solutions', 'positioning', 'acquisition', 'strategy', 'research', 'database', 'forecast', 'interest', 'prepared', 'merchant', 'document', 'feedback', 'template', 'campaign', 'partners', 'settings', 'keystone', 'llm', 'gpt', 'chatgpt', 'openai', 'anthropic', 'claude', 'gemini', 'api', 'json', 'xml', 'html', 'css', 'javascript', 'python', 'golang', 'typescript', 'rust', 'fastapi', 'snowflake', 'kubernetes', 'terraform', 'docker', 'redis', 'kafka', 'pytorch', 'policy', 'terms', 'conditions', 'release', 'sprint', 'deployment', 'cluster', 'instance', 'package', 'module', 'revenue', 'margin', 'gross', 'quarter', 'system', 'code', 'data', 'cloud', 'server', 'database', 'artificial', 'intelligence', 'learning', 'generative', 'regular', 'hours', 'holiday', 'earnings', 'deductions', 'withheld', 'withholding', 'taxes', 'medicare', 'benefits', 'reimbursements', 'compensation', 'wages', 'code']);
380
464
 
381
465
  let NOT_NAME_WORDS = new Set([
382
466
  // Grammatical & Sentence Starters
383
- 'the', 'a', 'an', 'this', 'that', 'these', 'those', 'my', 'your', 'his', 'her', 'their', 'our', 'its', 'it', 'he', 'she', 'they', 'we', 'i', 'you', 'who', 'whom', 'which', 'what', 'whose', 'why', 'how', 'when', 'where', 'with', 'for', 'from', 'by', 'to', 'at', 'in', 'on', 'of', 'about', 'as', 'into', 'through', 'during', 'before', 'after', 'above', 'below', 'and', 'but', 'or', 'so', 'yet',
467
+ 'the', 'a', 'an', 'this', 'that', 'these', 'those', 'my', 'your', 'his', 'her', 'their', 'our', 'its', 'it', 'he', 'she', 'they', 'we', 'i', 'you', 'who', 'whom', 'which', 'what', 'whose', 'why', 'how', 'when', 'where', 'with', 'for', 'from', 'by', 'to', 'at', 'in', 'on', 'of', 'about', 'as', 'into', 'through', 'during', 'before', 'after', 'above', 'below', 'and', 'but', 'or', 'so', 'yet', 'im', "i'm", "you're", "they're", "we're", "it's", "he's", "she's", "that's", "there's", "what's", "who's", "i've", "you've", "we've", "they've", "i'll", "you'll", "we'll", "they'll", "i'd", "you'd", "we'd", "they'd",
468
+ // Verbs, Auxiliaries, Commands & Imperatives
469
+ 'do', 'does', 'did', 'done', 'doing', 'dont', "don't", 'doesnt', "doesn't", 'didnt', "didn't", 'not', 'no', 'never', 'always',
470
+ 'be', 'is', 'am', 'are', 'was', 'were', 'been', 'being',
471
+ 'have', 'has', 'had', 'having',
472
+ 'can', 'could', 'may', 'might', 'must', 'shall', 'should', 'will', 'would', 'wont', "won't", 'wouldnt', "wouldn't", 'shouldnt', "shouldn't", 'couldnt', "couldn't", 'cant', "can't", 'cannot',
473
+ 'write', 'writing', 'written', 'writes', 'read', 'reading', 'reads',
474
+ 'wait', 'waiting', 'waited', 'waits', 'place', 'placing', 'placed', 'places',
475
+ 'display', 'displaying', 'displayed', 'displays',
476
+ 'provide', 'providing', 'provided', 'provides',
477
+ 'show', 'showing', 'shown', 'shows',
478
+ 'tell', 'telling', 'told', 'tells',
479
+ 'ask', 'asking', 'asked', 'asks',
480
+ 'use', 'using', 'used', 'uses',
481
+ 'select', 'selecting', 'selected', 'selects',
482
+ 'find', 'finding', 'findings', 'found', 'finds',
483
+ 'reference', 'referencing', 'referenced', 'references',
484
+ 'access', 'accessing', 'accessed', 'accesses',
485
+ 'note', 'noting', 'noted', 'notes',
486
+ 'get', 'getting', 'got', 'gotten', 'gets',
487
+ 'make', 'making', 'made', 'makes',
488
+ 'give', 'giving', 'given', 'gives',
489
+ 'take', 'taking', 'took', 'taken', 'takes',
490
+ 'put', 'putting', 'puts',
491
+ 'set', 'setting', 'sets',
492
+ 'keep', 'keeping', 'kept', 'keeps',
493
+ 'let', 'letting', 'lets',
494
+ 'leave', 'leaving', 'left', 'leaves',
495
+ 'run', 'running', 'ran', 'runs',
496
+ 'stop', 'stopping', 'stopped', 'stops',
497
+ 'start', 'starting', 'started', 'starts',
498
+ 'check', 'checking', 'checked', 'checks',
499
+ 'print', 'printing', 'printed', 'prints',
500
+ 'generate', 'generating', 'generated', 'generates',
501
+ 'create', 'creating', 'created', 'creates',
502
+ 'build', 'building', 'built', 'builds',
503
+ 'include', 'including', 'included', 'includes',
504
+ 'exclude', 'excluding', 'excluded', 'excludes',
505
+ 'format', 'formatting', 'formatted', 'formats',
506
+ 'change', 'changing', 'changed', 'changes',
507
+ 'send', 'sending', 'sent', 'sends',
508
+ 'receive', 'receiving', 'received', 'receives',
509
+ 'delete', 'deleting', 'deleted', 'deletes',
510
+ 'remove', 'removing', 'removed', 'removes',
511
+ 'insert', 'inserting', 'inserted', 'inserts',
512
+ 'update', 'updating', 'updated', 'updates',
513
+ 'review', 'reviewing', 'reviewed', 'reviews',
514
+ 'allow', 'allowing', 'allowed', 'allows',
515
+ 'deny', 'denying', 'denied', 'denies',
516
+ 'require', 'requiring', 'required', 'requires',
517
+ 'turn', 'turning', 'turned', 'turns',
518
+ 'switch', 'switching', 'switched', 'switches',
519
+ 'enable', 'enabling', 'enabled', 'enables',
520
+ 'disable', 'disabling', 'disabled', 'disables',
521
+ 'ensure', 'ensuring', 'ensured', 'ensures',
522
+ 'verify', 'verifying', 'verified', 'verifies',
523
+ 'execute', 'executing', 'executed', 'executes',
524
+ 'test', 'testing', 'tested', 'tests',
525
+ 'install', 'installing', 'installed', 'installs',
526
+ 'uninstall', 'uninstalling', 'uninstalled', 'uninstalls',
527
+ 'suppose', 'supposed', 'supposing', 'supposes',
528
+ 'respond', 'responding', 'responded', 'responds',
529
+ 'preserve', 'preserving', 'preserved', 'preserves',
530
+ 'replace', 'replacing', 'replaced', 'replaces',
531
+ 'present', 'presented', 'presenting', 'presents',
532
+ 'admit', 'admitted', 'admitting', 'admits',
533
+ 'complain', 'complained', 'complaining', 'complains',
534
+ 'prescribe', 'prescribed', 'prescribing', 'prescribes',
535
+ 'diagnose', 'diagnosed', 'diagnosing', 'diagnoses',
536
+ 'report', 'reported', 'reporting', 'reports',
537
+ 'state', 'stated', 'stating', 'states',
538
+ 'undergo', 'underwent', 'undergoing', 'undergoes',
539
+ 'experience', 'experienced', 'experiencing', 'experiences',
540
+ 'arrive', 'arrived', 'arriving', 'arrives',
541
+ 'order', 'ordered', 'ordering', 'orders',
542
+ // Adverbs, Prepositions, Conjunctions & Modifiers
543
+ 'again', 'without', 'with', 'within', 'specific', 'specifically', 'permission', 'permissions',
544
+ 'underneath', 'above', 'below', 'between', 'among', 'together', 'separately', 'instead',
545
+ 'also', 'too', 'either', 'neither', 'both', 'each', 'every', 'all', 'some', 'any', 'none',
546
+ 'only', 'just', 'already', 'currently', 'more', 'most', 'less', 'least',
547
+ 'very', 'quite', 'rather', 'such', 'same', 'different', 'other', 'others', 'another',
548
+ 'like', 'unlike', 'similar', 'complete', 'completely', 'entire', 'entirely',
549
+ 'exact', 'exactly', 'approximate', 'approximately', 'general', 'generally',
550
+ 'direct', 'directly', 'indirect', 'indirectly', 'total', 'totally', 'full', 'fully',
551
+ 'partial', 'partially', 'own', 'proper', 'properly',
552
+ 'now', 'then', 'soon', 'later', 'here', 'there', 'everywhere', 'nowhere', 'somewhere', 'anywhere',
553
+ 'inside', 'outside', 'before', 'after', 'since', 'until', 'till',
554
+ 'while', 'whereas', 'unless', 'although', 'though', 'even', 'because',
555
+ 'therefore', 'however', 'furthermore', 'moreover', 'meanwhile', 'otherwise', 'besides', 'further',
384
556
  // Greetings & Salutations
385
557
  'hello', 'hi', 'hey', 'dear', 'greetings',
386
558
  // Document & Resume Structure
387
- 'summary', 'experience', 'education', 'skills', 'languages', 'project', 'history', 'background', 'objective', 'profile', 'awards', 'honors', 'certifications', 'publications', 'interests', 'references',
559
+ 'summary', 'experience', 'education', 'skills', 'languages', 'project', 'history', 'background', 'objective', 'profile', 'awards', 'honors', 'certifications', 'publications', 'interests', 'references', 'statement', 'statements', 'form', 'forms',
388
560
  // Business & Job Roles
389
561
  'manager', 'director', 'specialist', 'analyst', 'engineer', 'developer', 'consultant', 'officer', 'representative', 'agent', 'lead', 'leader', 'president', 'coordinator', 'admin', 'administrator', 'executive', 'founder', 'partner', 'intern', 'trainee', 'advisor', 'head', 'vp', 'chief',
390
562
  // Departments & Fields
391
563
  'marketing', 'sales', 'engineering', 'finance', 'accounting', 'legal', 'operations', 'support', 'recruiting', 'talent', 'acquisition', 'compliance', 'security', 'technical', 'development', 'product', 'design', 'creative', 'strategy', 'planning', 'analytics', 'science', 'business', 'administration',
392
564
  // Tools & Tech Concepts
393
- 'google', 'ads', 'analytics', 'meta', 'hubspot', 'crm', 'salesforce', 'wordpress', 'mailchimp', 'adobe', 'figma', 'canva', 'slack', 'zoom', 'teams', 'microsoft', 'office', 'excel', 'word', 'powerpoint', 'notion', 'jira', 'confluence', 'github', 'gitlab', 'aws', 'azure', 'cloud', 'database', 'sql', 'python', 'java', 'javascript', 'html', 'css', 'react', 'node', 'api', 'saas', 'b2b', 'b2c', 'url', 'domain', 'website', 'app', 'application', 'software', 'email', 'phone', 'contact', 'address',
565
+ 'google', 'ads', 'analytics', 'meta', 'hubspot', 'crm', 'salesforce', 'wordpress', 'mailchimp', 'adobe', 'figma', 'canva', 'slack', 'zoom', 'teams', 'microsoft', 'office', 'excel', 'word', 'powerpoint', 'notion', 'jira', 'confluence', 'github', 'gitlab', 'aws', 'gcp', 'azure', 'cloud', 'database', 'sql', 'python', 'golang', 'typescript', 'rust', 'fastapi', 'snowflake', 'kubernetes', 'terraform', 'docker', 'redis', 'kafka', 'pytorch', 'java', 'javascript', 'html', 'css', 'react', 'node', 'api', 'saas', 'b2b', 'b2c', 'url', 'domain', 'website', 'app', 'application', 'software', 'email', 'phone', 'contact', 'address',
394
566
 
395
567
  // General Academic & Professional vocabulary
396
568
  'bachelor', 'master', 'doctor', 'associate', 'degree', 'university', 'college', 'school', 'institute', 'academy', 'graduated', 'major', 'minor', 'gpa', 'cum', 'laude', 'honors', 'deans', 'list', 'scholarship',
@@ -399,15 +571,29 @@ let NOT_NAME_WORDS = new Set([
399
571
  // Common Resume / Business Phrases
400
572
  'results-driven', 'data-driven', 'customer-centric', 'detail-oriented', 'cross-functional', 'self-motivated', 'time-management', 'problem-solving', 'fast-paced', 'year-over-year',
401
573
  // Legal & Trust terms
402
- 'trust', 'trustee', 'co-trustee', 'settlor', 'grantor', 'beneficiary', 'agreement', 'will', 'estate', 'witness', 'declaration', 'signatory', 'testator', 'notary', 'commission', 'county', 'state', 'court', 'article', 'section', 'paragraph', 'schedule', 'exhibit', 'amendment', 'addendum', 'power', 'attorney', 'guardian', 'executor', 'administrator', 'survivor', 'predecessor', 'successor', 'whereof', 'hereby', 'thereby', 'herein', 'therein', 'witnesseth', 'whereas', 'therefore', 'now', 'dated', 'effective',
574
+ 'trust', 'trustee', 'co-trustee', 'settlor', 'grantor', 'beneficiary', 'agreement', 'will', 'estate', 'witness', 'declaration', 'signatory', 'testator', 'notary', 'commission', 'county', 'state', 'court', 'article', 'section', 'paragraph', 'schedule', 'exhibit', 'amendment', 'addendum', 'power', 'attorney', 'guardian', 'executor', 'administrator', 'survivor', 'predecessor', 'successor', 'whereof', 'hereby', 'thereby', 'herein', 'therein', 'witnesseth', 'whereas', 'therefore', 'now', 'dated', 'effective', 'matter', 'case', 'cases', 'docket', 'dockets', 'number', 'numbers', 'firm', 'firms', 'lawyer', 'lawyers', 'counsel', 'counsels', 'counselstack', 'tier', 'tiers', 'finding', 'findings', 'connector', 'connectors', 'platform', 'platforms',
575
+ // Medical & Clinical terms
576
+ 'clinical', 'note', 'notes', 'dx', 'rx', 'tx', 'hx', 'px', 'sx', 'type', 'diabetes', 'referred', 'referral', 'diagnosed', 'diagnosis', 'patient', 'insurance', 'bcbs', 'mrn', 'dob',
577
+ // Tax & Payroll terms
578
+ 'wages', 'wage', 'tips', 'compensation', 'withheld', 'withholding', 'medicare', 'deductions', 'deduction', 'earning', 'earnings', 'gross', 'net', 'pay', 'payroll', 'paystub', 'taxable', 'exempt', 'allowance', 'allowances', 'regular', 'hours', 'holiday', 'overtime', 'commission', 'bonus', 'bonuses', 'records', 'record', 'statement', 'statements', 'rate', 'rates', 'current', 'ytd', 'benefits', 'taxable', 'pre-tax', 'post-tax', 'reimbursements', 'reimbursement', 'fica', 'oasdi', 'disability', 'unemployment', 'sui', 'sdi', 'std', 'ltd', 'exemptions', 'exemption', 'allowances', 'allowance', 'filing', 'status', 'single', 'married', 'head', 'household', 'advice', 'frequency', 'bi-weekly', 'biweekly', 'weekly', 'monthly', 'semi-monthly', 'direct', 'deposit', 'routing', 'box', 'boxes', 'code', 'control', 'omb', 'copy', 'instructions', 'information', 'deferred', 'adoption', 'statutory', 'third-party', 'sick', 'form', 'schedule', 'w-2', 'w2', 'w-4', 'w4', '1099', 'k-1', '1040', 'fed', 'med', 'fwt', 'swt', 'fed w/h', 'fed med', 'locality', 'state wages', 'state tax', 'local wages', 'local tax', 'allocated', 'nonqualified',
579
+ // Common Web, UI, Compliance, Document & AI Terms (Suppresses false-positive Name detection on headlines, buttons, and badges)
580
+ 'incident', 'incidents', 'critical', 'production', 'impacted', 'reported', 'details', 'vulnerability', 'vulnerabilities', 'host', 'types', 'type', 'leaked', 'leak', 'leaks', 'masked', 'mask', 'masking', 'leave', 'screen', 'screens', 'risk', 'risks', 'cluster', 'clusters', 'parameter', 'parameters', 'processing', 'process', 'processed', 'verified', 'verify', 'verification', 'playground', 'guide', 'guides', 'protection', 'protect', 'corporate', 'enterprise', 'log', 'logs', 'airplane', 'mode', 'zero', 'trust', 'top', 'data', 'live', 'scrubber', 'scrub', 'scrubbed', 'note', 'notes', 'secret', 'secrets', 'card', 'cards', 'raw', 'input', 'output', 'contains', 'contain', 'contained', 'platform', 'solutions', 'pricing', 'company', 'news', 'dashboard', 'add', 'chrome', 'sample', 'samples', 'try', 'terms', 'privacy', 'policy', 'policies', 'home', 'compliance', 'framework', 'frameworks', 'audit', 'audits', 'receipt', 'receipts', 'overview', 'explore', 'vectors', 'vector', 'standard', 'standards', 'status', 'preview', 'view', 'actions', 'action', 'button', 'buttons', 'option', 'options', 'general', 'specialized', 'custom', 'rule', 'rules', 'token', 'tokens', 'value', 'values', 'session', 'sessions', 'local', 'server', 'servers', 'cloud', 'ram', 'memory', 'offline', 'online', 'client', 'browser', 'extension', 'workspace', 'workplace', 'pan', 'phi', 'pii', 'soc', 'soc2', 'gdpr', 'hipaa', 'ccpa', 'iso27001', 'pci', 'dss', 'nist', 'chatgpt', 'claude', 'gemini', 'copilot', 'perplexity', 'deepseek', 'qwen', 'grok', 'llama', 'mistral', 'ai', 'llm', 'prompt', 'prompts', 'transmission', 'transit', 'egress', 'neutralized', 'stripped', 'isolated', 'isolation', 'unlocked', 'locked', 'unlock', 'download', 'copy', 'dismiss', 'close', 'save', 'settings', 'protect', 'reveal', 'unmask', 'restore', 'restored', 'export', 'import',
581
+ // Games, Chess, and Playing Pieces
582
+ 'bishop', 'bishops', 'knight', 'knights', 'rook', 'rooks', 'pawn', 'pawns', 'king', 'kings', 'queen', 'queens', 'chessboard', 'checkmate', 'stalemate', 'castling', 'en passant', 'chess',
583
+ // Colors & Visual Descriptors
584
+ 'white', 'black', 'red', 'blue', 'green', 'yellow', 'orange', 'purple', 'pink', 'brown', 'gray', 'grey', 'dark', 'light', 'gold', 'silver', 'bronze',
585
+ // Animation, 3D Rendering & Prompt Terminology
586
+ 'pixar', 'disney', 'animation', 'render', 'rendering', 'composition', 'cinematic', 'smooth', 'glides', 'glide', 'gliding', 'capture', 'captures', 'capturing', 'camera', 'orbit', 'orbits', 'orbiting', 'trapped', 'trap', 'trapping', 'square', 'squares', 'character', 'characters', 'expressive', 'living', 'texture', 'textures', 'reflection', 'reflections', 'grain', 'candlelight', 'wooden', 'polished', 'vertical', 'horizontal', 'macro', 'closeup', 'close-up', 'scene', 'scenes', 'shot', 'shots', 'shadow', 'shadows',
587
+ // Email, Outreach, Guest Posting & Agency Business Vocabulary
588
+ 'guest', 'post', 'posts', 'posting', 'attached', 'attach', 'attachment', 'attachments', 'updated', 'update', 'updates', 'list', 'lists', 'line', 'lines', 'rate', 'rates', 'affordable', 'services', 'service', 'infotech', 'technologies', 'technology', 'agency', 'agencies', 'digital', 'marketing', 'traffic', 'smart', 'design', 'seo', 'per', 'host', 'hosting', 'sites', 'site', 'inbox', 'starred', 'snoozed', 'important', 'sent', 'drafts', 'draft', 'spam', 'bin', 'trash', 'purchases', 'travel', 'social', 'forums', 'promotions', 'promotion', 'reply', 'forward', 'labels', 'label', 'compose', 'message', 'messages', 'mailer', 'outreach', 'backlink', 'backlinks', 'domain', 'authority', 'da', 'dr', 'founder', 'ceo', 'cto', 'cfo', 'coo', 'vp', 'head', 'lead',
403
589
  // US States
404
- 'california', 'texas', 'florida', 'york', 'illinois', 'pennsylvania', 'ohio', 'georgia', 'michigan', 'carolina', 'virginia', 'washington', 'arizona', 'massachusetts', 'tennessee', 'indiana', 'maryland', 'missouri', 'wisconsin', 'colorado', 'minnesota', 'alabama', 'louisiana', 'kentucky', 'oregon', 'oklahoma', 'connecticut', 'utah', 'iowa', 'nevada', 'arkansas', 'mississippi', 'kansas', 'new mexico', 'nebraska', 'idaho', 'hawaii', 'maine', 'new hampshire', 'rhode island', 'montana', 'delaware', 'south dakota', 'north dakota', 'alaska', 'vermont', 'wyoming'
405
- , 'f.3d', 'f.supp', 'u.s.c.', 'v.', 'plaintiff', 'defendant', 'v', 'u.s.', 'court', 'app.', 'reporter', 'cir.']);
590
+ 'california', 'texas', 'florida', 'york', 'illinois', 'pennsylvania', 'ohio', 'georgia', 'michigan', 'carolina', 'virginia', 'washington', 'arizona', 'massachusetts', 'tennessee', 'indiana', 'maryland', 'missouri', 'wisconsin', 'colorado', 'minnesota', 'alabama', 'louisiana', 'kentucky', 'oregon', 'oklahoma', 'connecticut', 'utah', 'iowa', 'nevada', 'arkansas', 'mississippi', 'kansas', 'new mexico', 'nebraska', 'idaho', 'hawaii', 'maine', 'new hampshire', 'rhode island', 'montana', 'delaware', 'south dakota', 'north dakota', 'alaska', 'vermont', 'wyoming',
591
+ 'f.3d', 'f.supp', 'u.s.c.', 'v.', 'plaintiff', 'defendant', 'v', 'u.s.', 'court', 'app.', 'reporter', 'cir.']);
406
592
 
407
593
  const PROFILE_JARGON = {
408
- medical: ['sleep', 'apnea', 'symptom', 'symptoms', 'trauma', 'hypertension', 'health', 'disease', 'condition', 'diagnosis', 'treatment', 'medication', 'dose', 'patient', 'clinic', 'surgery', 'therapy', 'alcohol', 'cannabis', 'blood', 'pressure', 'heart', 'rate', 'emergency', 'contact', 'relationship'],
594
+ medical: ['sleep', 'apnea', 'symptom', 'symptoms', 'trauma', 'hypertension', 'health', 'disease', 'condition', 'diagnosis', 'treatment', 'medication', 'dose', 'patient', 'clinic', 'surgery', 'therapy', 'alcohol', 'cannabis', 'blood', 'pressure', 'heart', 'rate', 'emergency', 'contact', 'relationship', 'type', 'diabetes', 'cancer', 'asthma', 'copd', 'covid', 'infection', 'syndrome', 'disorder', 'chronic', 'acute', 'illness', 'fever', 'allergy', 'pain', 'referral', 'referred', 'prescription', 'prescribed', 'doctor', 'physician', 'nurse', 'hospital', 'clinical', 'note', 'notes', 'dx', 'rx', 'tx', 'hx', 'px', 'sx', 'insurance', 'bcbs', 'bronchitis', 'amoxicillin', 'penicillin', 'antibiotic', 'antibiotics', 'vital', 'vitals', 'bp', 'hr', 'bpm', 'mmhg', 'allergies', 'dosage'],
409
595
  realestate: ['escrow', 'tenant', 'landlord', 'lease', 'mortgage', 'appraisal', 'broker', 'property', 'zoning', 'parcel', 'rent', 'buyer', 'seller', 'agent', 'listing'],
410
- legal: ['testator', 'notary', 'commission', 'county', 'court', 'affidavit', 'plaintiff', 'defendant', 'litigation', 'jurisdiction', 'agreement', 'contract', 'settlement', 'clause', 'article', 'section'],
596
+ legal: ['testator', 'notary', 'commission', 'county', 'court', 'affidavit', 'plaintiff', 'defendant', 'litigation', 'jurisdiction', 'agreement', 'contract', 'settlement', 'clause', 'article', 'section', 'matter', 'case'],
411
597
  hr: ['candidate', 'employee', 'payroll', 'benefits', 'salary', 'vacation', 'supervisor', 'subordinate', 'performance', 'appraisal', 'interview', 'resume', 'applicant'],
412
598
  sales: ['prospect', 'opportunity', 'quota', 'pipeline', 'deal', 'revenue', 'forecast', 'lead', 'churn', 'client', 'customer']
413
599
  };
@@ -447,11 +633,14 @@ const PROFILE_JARGON = {
447
633
  }
448
634
 
449
635
  const PROFILE_ALIAS_MAP = {
636
+ 'general': 'general',
637
+ 'underwriting': 'underwriting', 'lending': 'underwriting', 'mortgage': 'underwriting', 'loan': 'underwriting', 'income': 'underwriting', 'income_verification': 'underwriting', 'payroll': 'underwriting', 'w2': 'underwriting', 'paystub': 'underwriting',
450
638
  'medical': 'medical', 'healthcare': 'medical', 'health': 'medical', 'pharma': 'pharma',
451
- 'engineering': 'engineering', 'dev': 'engineering', 'tech': 'tech',
452
- 'finance': 'finance', 'bizops': 'bizops', 'sales': 'sales', 'wealthmgmt': 'wealthmgmt', 'insurance': 'insurance', 'accounting': 'accounting',
639
+ 'engineering': 'engineering', 'dev': 'engineering', 'devops': 'engineering', 'tech': 'tech',
640
+ 'finance': 'finance', 'bizops': 'bizops', 'sales': 'sales', 'wealthmgmt': 'wealthmgmt', 'wealth': 'wealthmgmt', 'insurance': 'insurance', 'accounting': 'accounting',
453
641
  'legal': 'legal', 'compliance': 'compliance', 'ccpa': 'ccpa',
454
642
  'hr': 'hr', 'security': 'security', 'marketing': 'marketing', 'support': 'support',
643
+ 'realestate': 'realestate', 'academic': 'academic', 'agents': 'agents', 'ai_agents': 'agents', 'creative': 'creative', 'personal': 'personal'
455
644
  };
456
645
 
457
646
  function getActiveRules(activeProfile) {
@@ -459,7 +648,7 @@ const PROFILE_JARGON = {
459
648
  if (activeProfile && activeProfile.toLowerCase() !== 'general') {
460
649
  const canonicalProfile = PROFILE_ALIAS_MAP[activeProfile.toLowerCase()] || 'general';
461
650
  if (canonicalProfile !== 'general' && PROFILE_RULES[canonicalProfile]) {
462
- activeRules = activeRules.concat(PROFILE_RULES[canonicalProfile]);
651
+ activeRules = PROFILE_RULES[canonicalProfile].concat(activeRules);
463
652
  }
464
653
  }
465
654
  return activeRules;
@@ -482,14 +671,15 @@ const PROFILE_JARGON = {
482
671
 
483
672
  if (customRules && customRules.length > 0) {
484
673
  const sorted = [...customRules].sort((a, b) => {
485
- const patternA = typeof a === 'string' ? a : a.pattern;
486
- const patternB = typeof b === 'string' ? b : b.pattern;
674
+ const patternA = typeof a === 'string' ? a : (a.pattern || (a.regex ? a.regex.source : '') || '');
675
+ const patternB = typeof b === 'string' ? b : (b.pattern || (b.regex ? b.regex.source : '') || '');
487
676
  return patternB.length - patternA.length;
488
677
  });
489
678
 
490
679
  sorted.forEach(cr => {
491
- const pattern = typeof cr === 'string' ? cr : cr.pattern;
492
- const label = typeof cr === 'string' ? 'CUSTOM' : (cr.label || 'CUSTOM');
680
+ const pattern = typeof cr === 'string' ? cr : (cr.pattern || (cr.regex ? cr.regex.source : ''));
681
+ if (!pattern) return;
682
+ const label = typeof cr === 'string' ? 'CUSTOM' : (cr.label || cr.mask || cr.name || 'CUSTOM');
493
683
 
494
684
  let rx;
495
685
  try {
@@ -525,33 +715,56 @@ const PROFILE_JARGON = {
525
715
  let matchedText = m[0];
526
716
  let start = m.index;
527
717
 
528
- if (rule.isContextName && m[1]) {
718
+ if (m.length > 1 && m[1] !== undefined && m[1] !== '') {
529
719
  matchedText = m[1];
530
- start = m.index + m[0].lastIndexOf(m[1]);
720
+ const relOffset = m[0].indexOf(m[1]);
721
+ if (relOffset !== -1) {
722
+ start = m.index + relOffset;
723
+ }
531
724
  }
532
725
  const end = start + matchedText.length;
533
726
 
534
727
  if (rule.type !== 'NAME' && rule.type !== 'ADDRESS') {
728
+ const val = matchedText.toLowerCase().trim();
729
+ if (NAME_STOP_LIST.has(val) || NOT_NAME_WORDS.has(val)) {
730
+ // Skip if generic English dictionary term matched by greedy regex (e.g. SWIFT matching SCRUBBER or CONTAINS)
731
+ if (rule.type === 'FINANCIAL' || rule.type === 'ID' || rule.type === 'PRIVACY' || rule.type === 'SECRET') {
732
+ continue;
733
+ }
734
+ }
535
735
  matches.push({ start, end, value: matchedText, type: rule.type });
536
736
  } else {
737
+ if (rule.type === 'NAME') {
738
+ matchedText = matchedText.replace(/[.,;:]+$/, '').trim();
739
+ }
537
740
  let val = matchedText.toLowerCase().trim();
538
- if (NAME_STOP_LIST.has(val)) continue;
741
+ if (!val || NAME_STOP_LIST.has(val)) continue;
539
742
 
540
- if (!rule.isContextName && rule.type === 'NAME') {
743
+ if (rule.type === 'NAME') {
541
744
  let words = val.split(/[ \t\xA0]+/);
542
- let origWords = matchedText.split(/[ \t\xA0]+/);
543
- while (words.length > 2 && (currentJargon.has(words[0]) || NOT_NAME_WORDS.has(words[0]))) {
544
- origWords.shift();
545
- words.shift();
546
- const nextStart = matchedText.indexOf(origWords[0]);
547
- start += nextStart;
548
- matchedText = matchedText.substring(nextStart);
549
- val = matchedText.toLowerCase().trim();
745
+ if (!rule.isContextName) {
746
+ let origWords = matchedText.split(/[ \t\xA0]+/);
747
+ while (words.length > 2 && (currentJargon.has(words[0]) || (words[0].length > 1 && NOT_NAME_WORDS.has(words[0])) || (words[0].replace(/[^\p{L}]/gu, '').length > 1 && NOT_NAME_WORDS.has(words[0].replace(/[^\p{L}]/gu, ''))))) {
748
+ origWords.shift();
749
+ words.shift();
750
+ const nextStart = matchedText.indexOf(origWords[0]);
751
+ if (nextStart !== -1) {
752
+ start += nextStart;
753
+ matchedText = matchedText.substring(nextStart);
754
+ val = matchedText.toLowerCase().trim();
755
+ } else {
756
+ break;
757
+ }
758
+ }
550
759
  }
551
- if (words.some(w => currentJargon.has(w) || NOT_NAME_WORDS.has(w))) continue;
760
+ if (!rule.isContextName && words.some(w => {
761
+ const cleanW = w.replace(/[^\p{L}]/gu, '');
762
+ if (cleanW.length <= 1) return false;
763
+ return currentJargon.has(w) || NOT_NAME_WORDS.has(w) || NOT_NAME_WORDS.has(cleanW) || NAME_STOP_LIST.has(w) || NAME_STOP_LIST.has(cleanW);
764
+ })) continue;
552
765
  } else if (rule.type === 'ADDRESS') {
553
- const words = val.split(/\s+/);
554
- if (words.some(w => currentJargon.has(w))) continue;
766
+ const cleanVal = val.replace(/[.,;!?]/g, ' ').trim();
767
+ if (NAME_STOP_LIST.has(cleanVal) || currentJargon.has(cleanVal)) continue;
555
768
  }
556
769
 
557
770
  matches.push({ start, end: start + matchedText.length, value: matchedText, type: rule.type });
@@ -588,7 +801,7 @@ const PROFILE_JARGON = {
588
801
  nameWords.forEach(w => {
589
802
  if (w && w.length >= 2 && /^\p{Lu}/u.test(w)) {
590
803
  const wl = w.toLowerCase();
591
- if (!NOT_NAME_WORDS.has(wl) && !JARGON_WORDS.has(wl) && !NAME_STOP_LIST.has(wl)) {
804
+ if (!NOT_NAME_WORDS.has(wl) && !JARGON_WORDS.has(wl) && !NAME_STOP_LIST.has(wl) && !currentJargon.has(wl)) {
592
805
  learnedNames.add(w);
593
806
  }
594
807
  }
@@ -641,13 +854,24 @@ const PROFILE_JARGON = {
641
854
  return Array.from(aliases);
642
855
  }
643
856
 
857
+ function formatToken(label, index, format = 'brackets') {
858
+ const cleanLabel = String(label || 'PII').replace(/[^A-Za-z0-9_]/g, '_').toUpperCase();
859
+ switch(format) {
860
+ case 'xml': return `<${cleanLabel}_${index}>`;
861
+ case 'mustache': return `{{${cleanLabel}_${index}}}`;
862
+ case 'underscores': return `__${cleanLabel}_${index}__`;
863
+ case 'brackets':
864
+ default: return `[${cleanLabel}_${index}]`;
865
+ }
866
+ }
867
+
644
868
  function buildRestorationRegexAndRules(tokenMap) {
645
869
  const ObjectKeys = Object.keys(tokenMap);
646
870
  if (ObjectKeys.length === 0) return { compositeRegex: null, looseRules: [] };
647
871
 
648
872
  const sortedKeys = [...ObjectKeys].sort((a, b) => {
649
- const innerA = a.replace(/^\[|\]$/g, '');
650
- const innerB = b.replace(/^\[|\]$/g, '');
873
+ const innerA = a.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
874
+ const innerB = b.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
651
875
  const matchA = innerA.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
652
876
  const matchB = innerB.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
653
877
  if (matchA && matchB) {
@@ -665,7 +889,7 @@ const PROFILE_JARGON = {
665
889
  const regexParts = [];
666
890
 
667
891
  sortedKeys.forEach(k => {
668
- const inner = k.replace(/^\[|\]$/g, '');
892
+ const inner = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
669
893
  const match = inner.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
670
894
  if (match) {
671
895
  const label = match[1];
@@ -673,7 +897,8 @@ const PROFILE_JARGON = {
673
897
  const aliases = getLabelAliases(label);
674
898
 
675
899
  const escapedAliases = aliases.map(a => a.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
676
- const loosePattern = `\\[\\s*(?:${escapedAliases.join('|')})[-_\\s]*0*${baseIndex}\\s*\\]|(?<![A-Za-z0-9\\u0400-\\u04FF_])(?:${escapedAliases.join('|')})[-_\\s]*0*${baseIndex}(?![A-Za-z0-9\\u0400-\\u04FF_])`;
900
+ const aliasesGroup = `(?:${escapedAliases.join('|')})`;
901
+ const loosePattern = `(?:\\[\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*\\]|<\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*>|\\{\\{\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*\\}\\}|__\\s*${aliasesGroup}[-_\\s]*0*${baseIndex}\\s*__|(?<![A-Za-z0-9\\u0400-\\u04FF_])${aliasesGroup}[-_\\s]*0*${baseIndex}(?![A-Za-z0-9\\u0400-\\u04FF_]))(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})?`;
677
902
  looseRules.push({ token: k, pattern: loosePattern });
678
903
  }
679
904
  regexParts.push(k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
@@ -681,42 +906,157 @@ const PROFILE_JARGON = {
681
906
 
682
907
  let compositeRegex = null;
683
908
  if (regexParts.length > 0) {
684
- compositeRegex = new RegExp(`(?:\\b|\\[)?(?:(?:(?<!\\w)|(?<=\\s))(?:${regexParts.join('|')})(?:(?!\\w)|(?=\\s)))(?:\\b|\\])?`, 'g');
909
+ compositeRegex = new RegExp(`(?:\\b|\\[|<|\\{\\{|__)?(?:(?:(?<!\\w)|(?<=\\s))(?:${regexParts.join('|')})(?:(?!\\w)|(?=\\s)))(?:\\b|\\]|>|\\}\\}|__)?(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})?`, 'g');
685
910
  }
686
911
 
687
912
  return { compositeRegex, looseRules };
688
913
  }
689
914
 
915
+ function isJsonPayload(str) {
916
+ if (!str || typeof str !== "string") return false;
917
+ const trimmed = str.trim();
918
+ if (!((trimmed.startsWith("{") && trimmed.endsWith("}")) || (trimmed.startsWith("[") && trimmed.endsWith("]")))) {
919
+ return false;
920
+ }
921
+ try {
922
+ JSON.parse(trimmed);
923
+ return true;
924
+ } catch (_) {
925
+ return false;
926
+ }
927
+ }
928
+
690
929
  function cleanAIPromptPrefix(text) {
691
930
  if (!text) return "";
692
- let cleaned = text.replace(/^\s*(?:Claude responded|Claude|ChatGPT|Gemini|Grok|DeepSeek|Kimi|Copilot|Assistant|User)\s*(?::|\bsaid\b|\bresponded\b|(?=\s))\s*/i, "");
931
+ let cleaned = text;
932
+ // If text is a full valid JSON object or array, preserve structure
933
+ if (!isJsonPayload(cleaned)) {
934
+ // 1. Strip raw CSS / style blocks leaked from ChatGPT Canvas, web components or stylesheets (handles multi-line, unclosed and variable definitions)
935
+ cleaned = cleaned.replace(/^\s*(?:[.#][a-zA-Z0-9_-]+|\[[a-zA-Z0-9_#.:\-*>=,'"\s]+\]|:is\([^)]+\)|[a-zA-Z0-9_-]+)?\s*\{[^}]*?(?:\}\s*|\n\n+|$)/gi, "");
936
+ cleaned = cleaned.replace(/^[;{} \t\r\n]+/, "");
937
+ cleaned = cleaned.replace(/(?:^|\n)[a-zA-Z0-9_#.:\-*>[\]=\s,'"]+\{[^}]*(--[a-zA-Z0-9_-]+:|color-mix\(|var\()[^}]*\}/g, "");
938
+ }
939
+ // 2. Strip AI author prefixes and platform artifacts
940
+ cleaned = cleaned.replace(/^\s*(?:Claude responded|Claude|ChatGPT|Gemini|Grok|DeepSeek|Kimi|Copilot|Assistant|User)\s*(?::|\bsaid\b|\bresponded\b|(?=\s))\s*/i, "");
693
941
  cleaned = cleaned.replace(/^(?:Here (?:is|are) (?:the )?(?:redacted|scrubbed|sanitized|processed|clean|updated|modified) (?:text|output|version|data).*?[:\n]+|\*\*Scrubbed Text\*\*[:\n]+|### Scrubbed Text[:\n]+)/i, '');
694
942
  cleaned = cleaned.replace(/^\s*Edit\s*\n+/i, "");
695
943
  cleaned = cleaned.replace(/\s*\bEdit\s+in\s+a\s+page\b\s*$/i, "");
944
+ // 3. Strip stray leading colons, semicolons, or separators left by stripped icons/artifact headers
945
+ cleaned = cleaned.replace(/^[:;|\-\—\–]+(?=\n|$)/, "");
946
+ cleaned = cleaned.replace(/^[:;]+\s*/, "");
696
947
  return cleaned.trim();
697
948
  }
698
949
 
950
+ function buildFastTokenLookup(sessionMap) {
951
+ const lookup = new Map();
952
+ const customRegexParts = [];
953
+ const keys = Object.keys(sessionMap || {});
954
+
955
+ for (let i = 0; i < keys.length; i++) {
956
+ const k = keys[i];
957
+ const v = sessionMap[k];
958
+ lookup.set(k, v);
959
+ lookup.set(k.toUpperCase(), v);
960
+
961
+ const inner = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '');
962
+ const match = inner.match(/^([A-Za-z_0-9]+?)[-_]?(\d+)$/);
963
+ if (match) {
964
+ const label = match[1];
965
+ const baseIndex = parseInt(match[2], 10);
966
+ const aliases = getLabelAliases(label);
967
+ for (let a = 0; a < aliases.length; a++) {
968
+ const u = aliases[a].toUpperCase();
969
+ lookup.set(u + '_' + baseIndex, v);
970
+ lookup.set(u + '-' + baseIndex, v);
971
+ lookup.set(u + ' ' + baseIndex, v);
972
+ lookup.set(u + baseIndex, v);
973
+ lookup.set('[' + u + '_' + baseIndex + ']', v);
974
+ lookup.set('<' + u + '_' + baseIndex + '>', v);
975
+ lookup.set('{{' + u + '_' + baseIndex + '}}', v);
976
+ lookup.set('__' + u + '_' + baseIndex + '__', v);
977
+ lookup.set('[' + u + ' ' + baseIndex + ']', v);
978
+ lookup.set('[' + u + '-' + baseIndex + ']', v);
979
+ }
980
+ } else {
981
+ customRegexParts.push(k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
982
+ }
983
+ }
984
+
985
+ let regexStr = '(?:\\[\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*\\]|<\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*>|\\{\\{\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*\\}\\}|__\\s*[A-Za-z0-9_\\-А-Яа-яЁё ]+\\s*__|(?<=^|[^a-zA-Z0-9_А-Яа-яЁё])[A-Za-z_А-Яа-яЁё]+[-_\\s]*\\d+)';
986
+ if (customRegexParts.length > 0) {
987
+ regexStr = '(?:' + regexStr + '|' + customRegexParts.join('|') + ')';
988
+ }
989
+ const tokenRegex = new RegExp(regexStr + '(?:\'s|’s|s|[а-яёА-ЯЁ]{1,3})?', 'gi');
990
+
991
+ return { lookup, tokenRegex };
992
+ }
993
+
994
+ function resolveTokenValue(rawMatch, targetTokenKey, sessionMap) {
995
+ if (!sessionMap) return undefined;
996
+ if (sessionMap[targetTokenKey] !== undefined) return sessionMap[targetTokenKey];
997
+ if (sessionMap[rawMatch] !== undefined) return sessionMap[rawMatch];
998
+ const cleanRaw = rawMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
999
+ for (const k of Object.keys(sessionMap)) {
1000
+ const cleanK = k.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
1001
+ if (cleanK.toLowerCase() === cleanRaw.toLowerCase()) {
1002
+ return sessionMap[k];
1003
+ }
1004
+ }
1005
+ return undefined;
1006
+ }
1007
+
699
1008
  function unscrubText(text, sessionMap) {
700
1009
  let restoredCount = 0;
701
1010
  let result = text;
702
- const tokens = Object.keys(sessionMap);
1011
+ const tokens = Object.keys(sessionMap || {});
703
1012
  if (tokens.length === 0) return { text: result, count: 0 };
704
1013
 
705
1014
  result = cleanAIPromptPrefix(result);
706
1015
 
1016
+ if (tokens.length > 50) {
1017
+ const { lookup, tokenRegex } = buildFastTokenLookup(sessionMap);
1018
+ result = result.replace(tokenRegex, (match) => {
1019
+ if (lookup.has(match)) {
1020
+ restoredCount++;
1021
+ return lookup.get(match);
1022
+ }
1023
+ const upper = match.toUpperCase();
1024
+ if (lookup.has(upper)) {
1025
+ restoredCount++;
1026
+ return lookup.get(upper);
1027
+ }
1028
+ const possMatch = match.match(/^([\s\S]+?)('s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1029
+ if (possMatch) {
1030
+ const base = possMatch[1];
1031
+ const suffix = possMatch[2];
1032
+ if (lookup.has(base)) {
1033
+ restoredCount++;
1034
+ return lookup.get(base) + suffix;
1035
+ }
1036
+ if (lookup.has(base.toUpperCase())) {
1037
+ restoredCount++;
1038
+ return lookup.get(base.toUpperCase()) + suffix;
1039
+ }
1040
+ }
1041
+ return match;
1042
+ });
1043
+ return { text: result, count: restoredCount };
1044
+ }
1045
+
707
1046
  const { compositeRegex, looseRules } = buildRestorationRegexAndRules(sessionMap);
708
1047
 
709
1048
  if (compositeRegex) {
710
1049
  result = result.replace(compositeRegex, (match) => {
711
- const cleanMatch = match.replace(/^\[|\]$/g, '').trim();
712
- const token = `[${cleanMatch.replace(/^\[|\]$/g, '')}]`;
713
- if (sessionMap[token]) {
714
- restoredCount++;
715
- return sessionMap[token];
716
- }
717
- if (sessionMap[cleanMatch]) {
1050
+ const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1051
+ const suffix = suffixMatch ? suffixMatch[0] : '';
1052
+ const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
1053
+ const cleanMatch = baseMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
1054
+ const token = `[${cleanMatch}]`;
1055
+
1056
+ const val = resolveTokenValue(baseMatch, token, sessionMap) ?? resolveTokenValue(baseMatch, cleanMatch, sessionMap);
1057
+ if (val !== undefined) {
718
1058
  restoredCount++;
719
- return sessionMap[cleanMatch];
1059
+ return val + suffix;
720
1060
  }
721
1061
  return match;
722
1062
  });
@@ -725,8 +1065,16 @@ const PROFILE_JARGON = {
725
1065
  looseRules.forEach(rule => {
726
1066
  const rx = new RegExp(rule.pattern, 'gi');
727
1067
  result = result.replace(rx, (match) => {
728
- restoredCount++;
729
- return sessionMap[rule.token];
1068
+ const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1069
+ const suffix = suffixMatch ? suffixMatch[0] : '';
1070
+ const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
1071
+
1072
+ const val = sessionMap[rule.token] ?? resolveTokenValue(baseMatch, rule.token, sessionMap);
1073
+ if (val !== undefined) {
1074
+ restoredCount++;
1075
+ return val + suffix;
1076
+ }
1077
+ return match;
730
1078
  });
731
1079
  });
732
1080
 
@@ -750,21 +1098,50 @@ const PROFILE_JARGON = {
750
1098
  const tokens = Object.keys(sessionMap || {});
751
1099
  if (tokens.length === 0) return { text: result, count: 0 };
752
1100
 
1101
+ if (tokens.length > 50) {
1102
+ const { lookup, tokenRegex } = buildFastTokenLookup(sessionMap);
1103
+ result = result.replace(tokenRegex, (match) => {
1104
+ let rawVal = null;
1105
+ let suffix = '';
1106
+ if (lookup.has(match)) {
1107
+ rawVal = lookup.get(match);
1108
+ } else if (lookup.has(match.toUpperCase())) {
1109
+ rawVal = lookup.get(match.toUpperCase());
1110
+ } else {
1111
+ const possMatch = match.match(/^([\s\S]+?)('s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1112
+ if (possMatch) {
1113
+ const base = possMatch[1];
1114
+ suffix = possMatch[2];
1115
+ if (lookup.has(base)) rawVal = lookup.get(base);
1116
+ else if (lookup.has(base.toUpperCase())) rawVal = lookup.get(base.toUpperCase());
1117
+ }
1118
+ }
1119
+ if (rawVal !== null) {
1120
+ restoredCount++;
1121
+ const escapedVal = escapeHTML(rawVal);
1122
+ const cleanMatch = match.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
1123
+ return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(cleanMatch)}">${escapedVal}</span>${escapeHTML(suffix)}`;
1124
+ }
1125
+ return match;
1126
+ });
1127
+ return { text: result, count: restoredCount };
1128
+ }
1129
+
753
1130
  const { compositeRegex, looseRules } = buildRestorationRegexAndRules(sessionMap);
754
1131
 
755
1132
  if (compositeRegex) {
756
1133
  result = result.replace(compositeRegex, (match) => {
757
- const cleanMatch = match.replace(/^\[|\]$/g, '').trim();
758
- const token = `[${cleanMatch.replace(/^\[|\]$/g, '')}]`;
759
- if (sessionMap[token]) {
760
- restoredCount++;
761
- const escapedVal = escapeHTML(sessionMap[token]);
762
- return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(token)}">${escapedVal}</span>`;
763
- }
764
- if (sessionMap[cleanMatch]) {
1134
+ const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1135
+ const suffix = suffixMatch ? suffixMatch[0] : '';
1136
+ const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
1137
+ const cleanMatch = baseMatch.replace(/^\[|<|\{\{|__|\]|>|\}\}|__/g, '').trim();
1138
+ const token = `[${cleanMatch}]`;
1139
+
1140
+ const val = resolveTokenValue(baseMatch, token, sessionMap) ?? resolveTokenValue(baseMatch, cleanMatch, sessionMap);
1141
+ if (val !== undefined) {
765
1142
  restoredCount++;
766
- const escapedVal = escapeHTML(sessionMap[cleanMatch]);
767
- return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(cleanMatch)}">${escapedVal}</span>`;
1143
+ const escapedVal = escapeHTML(val);
1144
+ return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(token)}">${escapedVal}</span>${escapeHTML(suffix)}`;
768
1145
  }
769
1146
  return match;
770
1147
  });
@@ -778,9 +1155,17 @@ const PROFILE_JARGON = {
778
1155
  const lastClose = before.lastIndexOf('>');
779
1156
  if (lastOpen > lastClose) return match;
780
1157
 
781
- restoredCount++;
782
- const escapedVal = escapeHTML(sessionMap[rule.token]);
783
- return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(rule.token)} (Fuzzy Match)">${escapedVal}</span>`;
1158
+ const suffixMatch = match.match(/(?:'s|’s|s|[а-яёА-ЯЁ]{1,3})$/);
1159
+ const suffix = suffixMatch ? suffixMatch[0] : '';
1160
+ const baseMatch = suffix ? match.slice(0, -suffix.length) : match;
1161
+
1162
+ const val = sessionMap[rule.token] ?? resolveTokenValue(baseMatch, rule.token, sessionMap);
1163
+ if (val !== undefined) {
1164
+ restoredCount++;
1165
+ const escapedVal = escapeHTML(val);
1166
+ return `<span class="ps-restored-data entity-tag entity-revealed" title="Original Token: ${escapeHTML(rule.token)} (Fuzzy Match)">${escapedVal}</span>${escapeHTML(suffix)}`;
1167
+ }
1168
+ return match;
784
1169
  });
785
1170
  });
786
1171
 
@@ -798,6 +1183,7 @@ const PROFILE_JARGON = {
798
1183
  LABEL_ALIASES,
799
1184
  getLabelAliases,
800
1185
  PROFILE_ALIAS_MAP,
1186
+ formatToken,
801
1187
  buildRestorationRegexAndRules,
802
1188
  unscrubText,
803
1189
  unscrubTextAsHTML,