@stll/anonymize-wasm 1.4.10 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,151 @@
1
+ //#region src/data/names-nw-vi.json
2
+ var _comment = "Non-Western name tokens for vi locale";
3
+ var names = [
4
+ "Anh",
5
+ "Bao",
6
+ "Bich",
7
+ "Bind",
8
+ "Binh",
9
+ "Bui",
10
+ "Cao",
11
+ "Chanh",
12
+ "Chau",
13
+ "Chien",
14
+ "Chon",
15
+ "Chuan",
16
+ "Cuc",
17
+ "Cuong",
18
+ "Dang",
19
+ "Dao",
20
+ "Dat",
21
+ "Dau",
22
+ "Diep",
23
+ "Dinh",
24
+ "Doanh",
25
+ "Dok",
26
+ "Duc",
27
+ "Dung",
28
+ "Duong",
29
+ "Duyen",
30
+ "Giang",
31
+ "Giao",
32
+ "Hai",
33
+ "Han",
34
+ "Hang",
35
+ "Hanh",
36
+ "Hao",
37
+ "Hien",
38
+ "Hieu",
39
+ "Hoa",
40
+ "Hoang",
41
+ "Hong",
42
+ "Huong",
43
+ "Huu",
44
+ "Huy",
45
+ "Huynh",
46
+ "Khang",
47
+ "Khanh",
48
+ "Khoa",
49
+ "Khuyen",
50
+ "Kie",
51
+ "Kien",
52
+ "Kieu",
53
+ "Kim",
54
+ "Kiu",
55
+ "La",
56
+ "Lam",
57
+ "Lan",
58
+ "Lao",
59
+ "Le",
60
+ "Lien",
61
+ "Linh",
62
+ "Loan",
63
+ "Loc",
64
+ "Long",
65
+ "Luong",
66
+ "Luu",
67
+ "Ly",
68
+ "Mai",
69
+ "Manh",
70
+ "Minh",
71
+ "My",
72
+ "Nam",
73
+ "Nga",
74
+ "Nghia",
75
+ "Nghiem",
76
+ "Ngo",
77
+ "Ngoc",
78
+ "Nguyen",
79
+ "Nhan",
80
+ "Nhat",
81
+ "Nhi",
82
+ "Nhu",
83
+ "Nhung",
84
+ "Ninh",
85
+ "Nong",
86
+ "Oanh",
87
+ "Pham",
88
+ "Phan",
89
+ "Phat",
90
+ "Phong",
91
+ "Phu",
92
+ "Phuc",
93
+ "Phung",
94
+ "Phuong",
95
+ "Quach",
96
+ "Quan",
97
+ "Quang",
98
+ "Quy",
99
+ "Quyet",
100
+ "Quynh",
101
+ "Sang",
102
+ "Sen",
103
+ "Sy",
104
+ "Ta",
105
+ "Tan",
106
+ "Thanh",
107
+ "Thao",
108
+ "Thi",
109
+ "Thien",
110
+ "Thiet",
111
+ "Thip",
112
+ "Thu",
113
+ "Thuan",
114
+ "Thuc",
115
+ "Thuy",
116
+ "Tien",
117
+ "Tram",
118
+ "Tran",
119
+ "Trang",
120
+ "Trieu",
121
+ "Trinh",
122
+ "Trong",
123
+ "Truc",
124
+ "Trung",
125
+ "Truong",
126
+ "Tuan",
127
+ "Tuong",
128
+ "Tuyet",
129
+ "Ty",
130
+ "Ung",
131
+ "Uy",
132
+ "Uyen",
133
+ "Vien",
134
+ "Viet",
135
+ "Vinh",
136
+ "Vo",
137
+ "Vu",
138
+ "Vuong",
139
+ "Vy",
140
+ "Xuan",
141
+ "Yat",
142
+ "Yen"
143
+ ];
144
+ var names_nw_vi_default = {
145
+ _comment,
146
+ names
147
+ };
148
+ //#endregion
149
+ export { _comment, names_nw_vi_default as default, names };
150
+
151
+ //# sourceMappingURL=names-nw-vi.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"names-nw-vi.mjs","names":[],"sources":["../../src/data/names-nw-vi.json"],"sourcesContent":[""],"mappings":""}
@@ -0,0 +1,197 @@
1
+ //#region src/data/names-nw-zh-latn.json
2
+ var _comment = "Non-Western name tokens for zh-latn locale";
3
+ var names = [
4
+ "Ang",
5
+ "AuYeung",
6
+ "Cao",
7
+ "Chak",
8
+ "Cham",
9
+ "Chan",
10
+ "Chang",
11
+ "Chao",
12
+ "Chau",
13
+ "Cheak",
14
+ "Chee",
15
+ "Chek",
16
+ "Chen",
17
+ "Cheong",
18
+ "Cheung",
19
+ "Chia",
20
+ "Chien",
21
+ "Chik",
22
+ "Chin",
23
+ "Ching",
24
+ "Chiu",
25
+ "Chng",
26
+ "Chon",
27
+ "Chong",
28
+ "Chow",
29
+ "Choy",
30
+ "Chu",
31
+ "Chua",
32
+ "Chuan",
33
+ "Chui",
34
+ "Chun",
35
+ "Chung",
36
+ "Cu",
37
+ "Cua",
38
+ "Dai",
39
+ "Dong",
40
+ "Dy",
41
+ "Eah",
42
+ "Fong",
43
+ "Foo",
44
+ "Fuk",
45
+ "Fung",
46
+ "Goh",
47
+ "Guan",
48
+ "Hak",
49
+ "Han",
50
+ "Hang",
51
+ "Hao",
52
+ "Heng",
53
+ "Heung",
54
+ "Ho",
55
+ "Hock",
56
+ "Hoi",
57
+ "Hong",
58
+ "Hsu",
59
+ "Huang",
60
+ "Huat",
61
+ "Hui",
62
+ "Hung",
63
+ "Jia",
64
+ "Jiang",
65
+ "Jun",
66
+ "Kam",
67
+ "Kan",
68
+ "Kee",
69
+ "Kek",
70
+ "Keung",
71
+ "Khoo",
72
+ "Kian",
73
+ "Kiat",
74
+ "Kie",
75
+ "Kieu",
76
+ "Kiok",
77
+ "Kiu",
78
+ "Ko",
79
+ "Koh",
80
+ "Kong",
81
+ "Koo",
82
+ "Koon",
83
+ "Kowk",
84
+ "Kuk",
85
+ "Kwan",
86
+ "Kwok",
87
+ "Kwong",
88
+ "La",
89
+ "Lai",
90
+ "Lam",
91
+ "Lau",
92
+ "Lee",
93
+ "Leong",
94
+ "Leung",
95
+ "Li",
96
+ "Lim",
97
+ "Lin",
98
+ "Ling",
99
+ "Liu",
100
+ "Lo",
101
+ "Lok",
102
+ "Lone",
103
+ "Long",
104
+ "Loo",
105
+ "Luk",
106
+ "Lun",
107
+ "Ma",
108
+ "Mak",
109
+ "Mei",
110
+ "Ming",
111
+ "Mok",
112
+ "Mong",
113
+ "Nai",
114
+ "Ng",
115
+ "Ngai",
116
+ "Ooi",
117
+ "Oon",
118
+ "Pai",
119
+ "Pak",
120
+ "Pang",
121
+ "Peh",
122
+ "Phua",
123
+ "Ping",
124
+ "Poe",
125
+ "Poon",
126
+ "Pua",
127
+ "Puah",
128
+ "Quek",
129
+ "Ren",
130
+ "Sen",
131
+ "Seng",
132
+ "Seow",
133
+ "Sham",
134
+ "Shek",
135
+ "Shen",
136
+ "Shum",
137
+ "Situ",
138
+ "Siu",
139
+ "Soh",
140
+ "Song",
141
+ "Soon",
142
+ "Suen",
143
+ "Suk",
144
+ "Sy",
145
+ "Szeto",
146
+ "Tai",
147
+ "Tak",
148
+ "Tam",
149
+ "Tan",
150
+ "Tang",
151
+ "Tay",
152
+ "Tee",
153
+ "Teo",
154
+ "Tham",
155
+ "Thom",
156
+ "Tiah",
157
+ "Ting",
158
+ "Tiu",
159
+ "Tong",
160
+ "Tsai",
161
+ "Tse",
162
+ "Tsen",
163
+ "Tsui",
164
+ "Tung",
165
+ "Wai",
166
+ "Wan",
167
+ "Wang",
168
+ "Wee",
169
+ "Wing",
170
+ "Wong",
171
+ "Woo",
172
+ "Wu",
173
+ "Wun",
174
+ "Xie",
175
+ "Yam",
176
+ "Yang",
177
+ "Yap",
178
+ "Yeoh",
179
+ "Yeung",
180
+ "Yi",
181
+ "Yim",
182
+ "Ying",
183
+ "Yip",
184
+ "Yiu",
185
+ "Yuen",
186
+ "Yuk",
187
+ "Yum",
188
+ "Zhang"
189
+ ];
190
+ var names_nw_zh_latn_default = {
191
+ _comment,
192
+ names
193
+ };
194
+ //#endregion
195
+ export { _comment, names_nw_zh_latn_default as default, names };
196
+
197
+ //# sourceMappingURL=names-nw-zh-latn.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"names-nw-zh-latn.mjs","names":[],"sources":["../../src/data/names-nw-zh-latn.json"],"sourcesContent":[""],"mappings":""}
package/dist/wasm.d.mts CHANGED
@@ -392,7 +392,8 @@ type RegexMeta = {
392
392
  label: string;
393
393
  score: number;
394
394
  sourceDetail?: Entity["sourceDetail"]; /** Post-match stdnum validator for confirmation. */
395
- validator?: Validator;
395
+ validator?: Validator; /** Extract the identifier portion when context is part of the regex span. */
396
+ validatorInput?: (text: string) => string;
396
397
  };
397
398
  /** Flat pattern array for text-search. */
398
399
  declare const REGEX_PATTERNS: readonly string[];
@@ -582,7 +583,12 @@ type NameCorpusData = {
582
583
  * Contains the lowercase, dot-stripped form
583
584
  * (e.g., "dr", "smt", "atty"). */
584
585
  titleAbbreviations: ReadonlySet<string>;
585
- excludedWords: ReadonlySet<string>; /** Non-Western name tokens merged across all locales. */
586
+ excludedWords: ReadonlySet<string>;
587
+ /** Lowercased common English words. A name chain whose
588
+ * every token is a common word (e.g. "Loan Documents",
589
+ * where "Loan" coincides with a Vietnamese given name)
590
+ * is treated as a common-word phrase, not a person. */
591
+ commonWords: ReadonlySet<string>; /** Non-Western name tokens merged across all locales. */
586
592
  nonWesternNames: ReadonlySet<string>; /** All-caps acronyms excluded from name detection. */
587
593
  excludedAllCaps: ReadonlySet<string>; /** Raw arrays exposed for deny-list AC integration. */
588
594
  firstNamesList: readonly string[];
@@ -1110,17 +1116,45 @@ declare const prepareBatch: (tokenizer: Tokenizer, texts: string[], entities: st
1110
1116
  };
1111
1117
  //#endregion
1112
1118
  //#region src/util/chunker.d.ts
1119
+ /** A chunk paired with its start offset in the source text. */
1120
+ type ChunkSpan = {
1121
+ text: string;
1122
+ offset: number;
1123
+ };
1113
1124
  /**
1114
- * Split text into overlapping chunks for GLiNER's
1115
- * ~512 token context window. Character-based splitting
1116
- * (rough approximation of token limits).
1125
+ * Split text into overlapping chunks, each paired with its
1126
+ * exact start offset in the source text.
1117
1127
  *
1118
- * Tries to break at sentence boundaries when possible.
1128
+ * Carrying the offset out of the splitter is the robust way to
1129
+ * map chunk-local entity offsets back to document offsets:
1130
+ * downstream code never has to re-locate a chunk by content
1131
+ * search (which mis-locates when boilerplate repeats; see
1132
+ * computeChunkOffsets).
1133
+ *
1134
+ * Character-based splitting (rough token approximation for
1135
+ * GLiNER's ~512 token window); breaks at sentence boundaries
1136
+ * when possible.
1137
+ */
1138
+ declare const chunkTextWithOffsets: (text: string) => ChunkSpan[];
1139
+ /**
1140
+ * Split text into overlapping chunks for GLiNER's ~512 token
1141
+ * context window. Character-based splitting (rough token
1142
+ * approximation); breaks at sentence boundaries when possible.
1143
+ *
1144
+ * Prefer chunkTextWithOffsets when you also need each chunk's
1145
+ * document offset.
1119
1146
  */
1120
1147
  declare const chunkText: (text: string) => string[];
1121
1148
  /**
1122
- * Compute the byte offset of each chunk within the
1123
- * original document text.
1149
+ * Compute the start offset of each chunk within the original
1150
+ * document text by content search.
1151
+ *
1152
+ * @deprecated Re-locates each chunk with `indexOf`, which can
1153
+ * match the wrong position when identical content repeats in
1154
+ * the document (common in boilerplate-heavy legal text) and
1155
+ * then desyncs every subsequent offset. Use
1156
+ * `chunkTextWithOffsets`, which carries exact offsets out of
1157
+ * the splitter.
1124
1158
  */
1125
1159
  declare const computeChunkOffsets: (fullText: string, chunks: string[]) => number[];
1126
1160
  /**
@@ -1179,5 +1213,5 @@ declare const levenshtein: (rawA: string, rawB: string) => number;
1179
1213
  */
1180
1214
  declare const normalizeForSearch: (text: string) => string;
1181
1215
  //#endregion
1182
- export { type AnonymisationOperator, CURRENCY_PATTERN_META, type CountryCode, type CustomDenyListEntry, type CustomRegexPattern, DATE_PATTERN_META, DEFAULT_ENTITY_LABELS, DEFAULT_OPERATOR_CONFIG, DETECTION_SOURCES, DETECTOR_PRIORITY, type DefinitionPattern, type DenyListCategory, type DenyListData, type DetectionSource, type Dictionaries, type DictionaryMeta, type DocumentZone, type Entity, type EntityResult, type GazetteerData, type GazetteerEntry, type HotwordRule, type NameCorpusData, type NerInferenceFn, OPERATOR_REGISTRY, OPERATOR_TYPES, type OperatorConfig, type OperatorType, type PipelineConfig, type PipelineContext, type PipelineOptions, type PipelineSearchOptions, REGEX_META, REGEX_PATTERNS, REGIONS, type RawInferenceResult, type RedactionResult, type RegexMeta, type RegionId, type ReviewDecision, type ReviewedEntity, type TriggerExtension, type TriggerGroupConfig, type TriggerRule, type TriggerStrategy, type TriggerValidation, type UnifiedResult, type UnifiedSearchInstance, ZONE_SCORE_ADJUSTMENTS, type ZoneSpan, applyHotwordRules, applyZoneAdjustments, boostNearMissEntities, buildDenyList, buildGazetteerPatterns, buildLegalFormPatterns, buildPlaceholderMap, buildStreetTypePatterns, buildTriggerPatterns, buildUnifiedSearch, chunkText, classifyZones, computeChunkOffsets, corefKey, createPipelineContext, deanonymise, decodeSpans, decodeTokenSpans, detectNameCorpus, ensureDenyListData, exportRedactionKey, extractDefinedTerms, filterFalsePositives, findCoreferenceSpans, getCurrencyPatterns, getDatePatterns, getNameCorpusNonWesternNames, initAddressComponents, initHotwordRules, initNameCorpus, initZoneClassifier, levenshtein, mergeAndDedup, mergeChunkEntities, normalizeForSearch, prepareBatch, preparePipelineSearch, processAddressSeeds, processDenyListMatches, processGazetteerMatches, processLegalFormMatches, processRegexMatches, processTriggerMatches, propagateOrgNames, redactText, resolveCountries, resolveOperator, runPipeline, runUnifiedSearch, sanitizeEntities, tokenizeText, warmLegalRoleHeads };
1216
+ export { type AnonymisationOperator, CURRENCY_PATTERN_META, type ChunkSpan, type CountryCode, type CustomDenyListEntry, type CustomRegexPattern, DATE_PATTERN_META, DEFAULT_ENTITY_LABELS, DEFAULT_OPERATOR_CONFIG, DETECTION_SOURCES, DETECTOR_PRIORITY, type DefinitionPattern, type DenyListCategory, type DenyListData, type DetectionSource, type Dictionaries, type DictionaryMeta, type DocumentZone, type Entity, type EntityResult, type GazetteerData, type GazetteerEntry, type HotwordRule, type NameCorpusData, type NerInferenceFn, OPERATOR_REGISTRY, OPERATOR_TYPES, type OperatorConfig, type OperatorType, type PipelineConfig, type PipelineContext, type PipelineOptions, type PipelineSearchOptions, REGEX_META, REGEX_PATTERNS, REGIONS, type RawInferenceResult, type RedactionResult, type RegexMeta, type RegionId, type ReviewDecision, type ReviewedEntity, type TriggerExtension, type TriggerGroupConfig, type TriggerRule, type TriggerStrategy, type TriggerValidation, type UnifiedResult, type UnifiedSearchInstance, ZONE_SCORE_ADJUSTMENTS, type ZoneSpan, applyHotwordRules, applyZoneAdjustments, boostNearMissEntities, buildDenyList, buildGazetteerPatterns, buildLegalFormPatterns, buildPlaceholderMap, buildStreetTypePatterns, buildTriggerPatterns, buildUnifiedSearch, chunkText, chunkTextWithOffsets, classifyZones, computeChunkOffsets, corefKey, createPipelineContext, deanonymise, decodeSpans, decodeTokenSpans, detectNameCorpus, ensureDenyListData, exportRedactionKey, extractDefinedTerms, filterFalsePositives, findCoreferenceSpans, getCurrencyPatterns, getDatePatterns, getNameCorpusNonWesternNames, initAddressComponents, initHotwordRules, initNameCorpus, initZoneClassifier, levenshtein, mergeAndDedup, mergeChunkEntities, normalizeForSearch, prepareBatch, preparePipelineSearch, processAddressSeeds, processDenyListMatches, processGazetteerMatches, processLegalFormMatches, processRegexMatches, processTriggerMatches, propagateOrgNames, redactText, resolveCountries, resolveOperator, runPipeline, runUnifiedSearch, sanitizeEntities, tokenizeText, warmLegalRoleHeads };
1183
1217
  //# sourceMappingURL=wasm.d.mts.map