@stll/anonymize 2.0.0-alpha.1 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (156) hide show
  1. package/README.md +1 -1
  2. package/dist/index.d.mts +2 -2
  3. package/dist/index.mjs +1 -1
  4. package/dist/native-node.d.mts +18 -4
  5. package/dist/native-node.mjs +1 -1
  6. package/dist/native-node2.d.mts +2 -2
  7. package/dist/native-node2.mjs +106 -12731
  8. package/dist/native-node2.mjs.map +1 -1
  9. package/dist/native.d.mts +526 -714
  10. package/dist/native.mjs.map +1 -1
  11. package/dist/native2.d.mts +2 -2
  12. package/package.json +27 -13
  13. package/dist/address-boundaries.mjs +0 -197
  14. package/dist/address-boundaries.mjs.map +0 -1
  15. package/dist/address-jurisdiction-prefixes.mjs +0 -16
  16. package/dist/address-jurisdiction-prefixes.mjs.map +0 -1
  17. package/dist/address-stop-keywords.mjs +0 -148
  18. package/dist/address-stop-keywords.mjs.map +0 -1
  19. package/dist/address-stopwords.mjs +0 -84
  20. package/dist/address-stopwords.mjs.map +0 -1
  21. package/dist/address-unit-abbreviations.mjs +0 -15
  22. package/dist/address-unit-abbreviations.mjs.map +0 -1
  23. package/dist/allow-list.mjs +0 -196
  24. package/dist/allow-list.mjs.map +0 -1
  25. package/dist/clause-noun-heads.mjs +0 -79
  26. package/dist/clause-noun-heads.mjs.map +0 -1
  27. package/dist/common-words-en.mjs +0 -9887
  28. package/dist/common-words-en.mjs.map +0 -1
  29. package/dist/coreference-org-determiners.mjs +0 -19
  30. package/dist/coreference-org-determiners.mjs.map +0 -1
  31. package/dist/coreference.cs.mjs +0 -14
  32. package/dist/coreference.cs.mjs.map +0 -1
  33. package/dist/coreference.de.mjs +0 -14
  34. package/dist/coreference.de.mjs.map +0 -1
  35. package/dist/coreference.en.mjs +0 -14
  36. package/dist/coreference.en.mjs.map +0 -1
  37. package/dist/coreference.es.mjs +0 -22
  38. package/dist/coreference.es.mjs.map +0 -1
  39. package/dist/coreference.fr.mjs +0 -32
  40. package/dist/coreference.fr.mjs.map +0 -1
  41. package/dist/coreference.it.mjs +0 -27
  42. package/dist/coreference.it.mjs.map +0 -1
  43. package/dist/coreference.pl.mjs +0 -27
  44. package/dist/coreference.pl.mjs.map +0 -1
  45. package/dist/coreference.pt-br.mjs +0 -14
  46. package/dist/coreference.pt-br.mjs.map +0 -1
  47. package/dist/coreference.sk.mjs +0 -27
  48. package/dist/coreference.sk.mjs.map +0 -1
  49. package/dist/currencies.mjs +0 -231
  50. package/dist/currencies.mjs.map +0 -1
  51. package/dist/date-months.mjs +0 -618
  52. package/dist/date-months.mjs.map +0 -1
  53. package/dist/defined-term-heads.mjs +0 -15
  54. package/dist/defined-term-heads.mjs.map +0 -1
  55. package/dist/document-structure-headings.mjs +0 -90
  56. package/dist/document-structure-headings.mjs.map +0 -1
  57. package/dist/false-positive-shapes.mjs +0 -36
  58. package/dist/false-positive-shapes.mjs.map +0 -1
  59. package/dist/generic-roles.mjs +0 -244
  60. package/dist/generic-roles.mjs.map +0 -1
  61. package/dist/hotword-rules.mjs +0 -149
  62. package/dist/hotword-rules.mjs.map +0 -1
  63. package/dist/legal-form-leading-clauses.mjs +0 -23
  64. package/dist/legal-form-leading-clauses.mjs.map +0 -1
  65. package/dist/legal-forms.mjs +0 -2115
  66. package/dist/legal-forms.mjs.map +0 -1
  67. package/dist/legal-role-heads.cs.mjs +0 -48
  68. package/dist/legal-role-heads.cs.mjs.map +0 -1
  69. package/dist/legal-role-heads.de.mjs +0 -33
  70. package/dist/legal-role-heads.de.mjs.map +0 -1
  71. package/dist/legal-role-heads.en.mjs +0 -37
  72. package/dist/legal-role-heads.en.mjs.map +0 -1
  73. package/dist/legal-role-heads.es.mjs +0 -54
  74. package/dist/legal-role-heads.es.mjs.map +0 -1
  75. package/dist/legal-role-heads.fr.mjs +0 -72
  76. package/dist/legal-role-heads.fr.mjs.map +0 -1
  77. package/dist/legal-role-heads.it.mjs +0 -68
  78. package/dist/legal-role-heads.it.mjs.map +0 -1
  79. package/dist/legal-role-heads.pl.mjs +0 -84
  80. package/dist/legal-role-heads.pl.mjs.map +0 -1
  81. package/dist/legal-role-heads.pt-br.mjs +0 -63
  82. package/dist/legal-role-heads.pt-br.mjs.map +0 -1
  83. package/dist/legal-role-heads.sk.mjs +0 -80
  84. package/dist/legal-role-heads.sk.mjs.map +0 -1
  85. package/dist/manifest.mjs +0 -69
  86. package/dist/manifest.mjs.map +0 -1
  87. package/dist/names-exclusions.mjs +0 -223
  88. package/dist/names-exclusions.mjs.map +0 -1
  89. package/dist/names-first.mjs +0 -418
  90. package/dist/names-first.mjs.map +0 -1
  91. package/dist/names-nw-ar.mjs +0 -202
  92. package/dist/names-nw-ar.mjs.map +0 -1
  93. package/dist/names-nw-excluded-allcaps.mjs +0 -112
  94. package/dist/names-nw-excluded-allcaps.mjs.map +0 -1
  95. package/dist/names-nw-fil.mjs +0 -202
  96. package/dist/names-nw-fil.mjs.map +0 -1
  97. package/dist/names-nw-id.mjs +0 -210
  98. package/dist/names-nw-id.mjs.map +0 -1
  99. package/dist/names-nw-in.mjs +0 -526
  100. package/dist/names-nw-in.mjs.map +0 -1
  101. package/dist/names-nw-ja-latn.mjs +0 -260
  102. package/dist/names-nw-ja-latn.mjs.map +0 -1
  103. package/dist/names-nw-ko.mjs +0 -162
  104. package/dist/names-nw-ko.mjs.map +0 -1
  105. package/dist/names-nw-th.mjs +0 -188
  106. package/dist/names-nw-th.mjs.map +0 -1
  107. package/dist/names-nw-vi.mjs +0 -151
  108. package/dist/names-nw-vi.mjs.map +0 -1
  109. package/dist/names-nw-zh-latn.mjs +0 -197
  110. package/dist/names-nw-zh-latn.mjs.map +0 -1
  111. package/dist/names-surnames.mjs +0 -113
  112. package/dist/names-surnames.mjs.map +0 -1
  113. package/dist/names-title-tokens.mjs +0 -40
  114. package/dist/names-title-tokens.mjs.map +0 -1
  115. package/dist/organization-unit-heads.mjs +0 -20
  116. package/dist/organization-unit-heads.mjs.map +0 -1
  117. package/dist/person-stopwords.mjs +0 -211
  118. package/dist/person-stopwords.mjs.map +0 -1
  119. package/dist/section-headings.mjs +0 -64
  120. package/dist/section-headings.mjs.map +0 -1
  121. package/dist/sentence-verb-indicators.mjs +0 -232
  122. package/dist/sentence-verb-indicators.mjs.map +0 -1
  123. package/dist/signing-clauses.mjs +0 -102
  124. package/dist/signing-clauses.mjs.map +0 -1
  125. package/dist/stopwords.mjs +0 -9915
  126. package/dist/stopwords.mjs.map +0 -1
  127. package/dist/structural-single-cap-prefixes.mjs +0 -99
  128. package/dist/structural-single-cap-prefixes.mjs.map +0 -1
  129. package/dist/triggers.cs.mjs +0 -569
  130. package/dist/triggers.cs.mjs.map +0 -1
  131. package/dist/triggers.de.mjs +0 -139
  132. package/dist/triggers.de.mjs.map +0 -1
  133. package/dist/triggers.en.mjs +0 -119
  134. package/dist/triggers.en.mjs.map +0 -1
  135. package/dist/triggers.es.mjs +0 -96
  136. package/dist/triggers.es.mjs.map +0 -1
  137. package/dist/triggers.fr.mjs +0 -275
  138. package/dist/triggers.fr.mjs.map +0 -1
  139. package/dist/triggers.global.mjs +0 -79
  140. package/dist/triggers.global.mjs.map +0 -1
  141. package/dist/triggers.hu.mjs +0 -41
  142. package/dist/triggers.hu.mjs.map +0 -1
  143. package/dist/triggers.it.mjs +0 -74
  144. package/dist/triggers.it.mjs.map +0 -1
  145. package/dist/triggers.pl.mjs +0 -271
  146. package/dist/triggers.pl.mjs.map +0 -1
  147. package/dist/triggers.pt-br.mjs +0 -193
  148. package/dist/triggers.pt-br.mjs.map +0 -1
  149. package/dist/triggers.ro.mjs +0 -59
  150. package/dist/triggers.ro.mjs.map +0 -1
  151. package/dist/triggers.sk.mjs +0 -555
  152. package/dist/triggers.sk.mjs.map +0 -1
  153. package/dist/triggers.sv.mjs +0 -58
  154. package/dist/triggers.sv.mjs.map +0 -1
  155. package/dist/year-words.mjs +0 -62
  156. package/dist/year-words.mjs.map +0 -1
package/dist/native.d.mts CHANGED
@@ -1,93 +1,95 @@
1
1
  import { i as DetectionSource, n as DETECTION_SOURCES, o as OperatorType } from "./constants2.mjs";
2
- import { Validator } from "@stll/stdnum";
3
- import { TextSearch } from "@stll/text-search";
4
2
 
5
- //#region src/types.d.ts
3
+ //#region src/native-search-config.d.ts
6
4
  /**
7
- * Fields shared by every entity span in the source text.
5
+ * Structural type for the prepared static-search config the native binding
6
+ * consumes and the Rust assembler emits (`assembleStaticSearchConfigJson`).
7
+ *
8
+ * This config used to be built in TypeScript by `build-unified-search.ts`; that
9
+ * layer was retired in favor of the Rust assembler
10
+ * (`crates/anonymize-adapter-contract` `assemble_static_search_config`). The
11
+ * type now lives here as a pure, dependency-free description of the JSON the
12
+ * binding accepts on its `fromConfigJsonBytes` / prepare paths, so callers that
13
+ * hold a pre-assembled config keep a precise type without pulling in the
14
+ * deleted detector modules.
8
15
  */
9
- type EntityBase = {
16
+ type PatternSlice = {
10
17
  start: number;
11
18
  end: number;
19
+ };
20
+ type NativeSearchPatternKind = "literal" | "literal-with-options" | "regex" | "fuzzy";
21
+ type NativeSearchPattern = {
22
+ kind: NativeSearchPatternKind;
23
+ pattern: string;
24
+ distance?: number;
25
+ case_insensitive?: boolean;
26
+ whole_words?: boolean;
27
+ lazy?: boolean;
28
+ prefilter_any?: string[];
29
+ prefilter_case_insensitive?: boolean;
30
+ prefilter_regex?: string;
31
+ prefilter_window_bytes?: number;
32
+ prepared_artifact_policy?: "include" | "omit";
33
+ };
34
+ type NativeSearchOptions = {
35
+ literal_case_insensitive?: boolean;
36
+ literal_whole_words?: boolean;
37
+ regex_whole_words?: boolean;
38
+ regex_overlap_all?: boolean;
39
+ regex_artifact_policy?: "include" | "omit";
40
+ fuzzy_case_insensitive?: boolean;
41
+ fuzzy_whole_words?: boolean;
42
+ fuzzy_normalize_diacritics?: boolean;
43
+ };
44
+ type NativeRegexMatchMeta = {
12
45
  label: string;
13
- text: string;
14
46
  score: number;
15
- sourceDetail?: "custom-deny-list" | "custom-regex" | "gazetteer-extension";
16
- };
17
- /**
18
- * A PII entity span found by a primary detection layer
19
- * (regex, NER, legal forms, deny list, ...).
20
- */
21
- type DetectedEntity = EntityBase & {
22
- source: Exclude<DetectionSource, typeof DETECTION_SOURCES.COREFERENCE>;
47
+ source_detail?: string;
48
+ requires_validation?: boolean;
49
+ validator_id?: string;
50
+ validator_input?: string;
51
+ min_byte_length?: number;
23
52
  };
24
- /**
25
- * An alias mention of a previously detected entity: a
26
- * defined term ("the Seller") or a propagated bare
27
- * mention ("Acme" after "Acme Corp.").
28
- *
29
- * `corefSourceText` is required by construction, so an
30
- * alias cannot exist without the link back to its source
31
- * entity. Placeholder numbering reads it to give the
32
- * alias the same placeholder as the source. The link
33
- * travels with the entity instead of living in a
34
- * side-channel map that a producer could forget to
35
- * write — or that a later pass could clear.
36
- */
37
- type CorefAliasEntity = EntityBase & {
38
- source: typeof DETECTION_SOURCES.COREFERENCE; /** Full text of the source entity this alias refers to. */
39
- corefSourceText: string;
53
+ type NativeSigningPlaceGuardData = {
54
+ prefix_phrases: string[];
55
+ suffix_phrases: string[];
40
56
  };
41
- /**
42
- * A detected PII entity span in the source text.
43
- * Every detection layer produces these.
44
- */
45
- type Entity = DetectedEntity | CorefAliasEntity;
46
- /**
47
- * Entity after human review. Extends the base Entity
48
- * with a review decision.
49
- */
50
- type ReviewDecision = "confirmed" | "rejected" | "relabeled";
51
- type ReviewedEntity = Entity & {
52
- decision?: ReviewDecision;
53
- originalLabel?: string;
57
+ type NativeDenyListFilterData = {
58
+ stopwords: string[];
59
+ allow_list: string[];
60
+ person_stopwords: string[];
61
+ person_trailing_nouns: string[];
62
+ address_stopwords: string[];
63
+ address_jurisdiction_prefixes: string[];
64
+ street_types: string[];
65
+ address_component_terms: string[];
66
+ ambiguous_street_type_terms: string[];
67
+ first_names: string[];
68
+ generic_roles: string[];
69
+ number_abbrev_prefixes: string[];
70
+ sentence_starters: string[];
71
+ trailing_address_word_exclusions: string[];
72
+ document_heading_words: string[];
73
+ document_heading_ordinal_markers: string[];
74
+ defined_term_cues: string[];
75
+ signing_place_guards: NativeSigningPlaceGuardData[];
54
76
  };
55
- /**
56
- * A single entry in the workspace-scoped gazetteer
57
- * (deny list). Persisted in IndexedDB.
58
- */
59
- type GazetteerEntry = {
60
- id: string;
61
- canonical: string;
62
- label: string;
63
- variants: string[];
64
- workspaceId: string;
65
- createdAt: number;
66
- source: "manual" | "confirmed-from-model";
77
+ type NativeDenyListMatchData = {
78
+ labels?: string[][];
79
+ label_table?: string[];
80
+ label_indices?: number[][];
81
+ custom_labels?: string[][];
82
+ custom_label_indices?: number[][];
83
+ originals: string[];
84
+ sources?: string[][];
85
+ source_table?: string[];
86
+ source_indices?: number[][];
87
+ filters?: NativeDenyListFilterData;
67
88
  };
68
- /** Extraction strategy — closed discriminated union. */
69
- type TriggerStrategy = {
89
+ type NativeTriggerStrategy = {
70
90
  type: "to-next-comma";
71
- /**
72
- * Optional list of lowercase keywords that terminate
73
- * the value scan, in addition to commas/newlines. Useful
74
- * for triggers like court names that may continue past
75
- * a missing comma into adjacent clause text ("Městským
76
- * soudem v Praze dne 1. 1. 2020"); listing `"dne"` here
77
- * stops the scan at the date boundary. Matched on a
78
- * word-boundary, case-insensitive.
79
- */
80
- stopWords?: string[];
81
- /**
82
- * Hard cap on the captured span length, in characters,
83
- * regardless of where the next comma / stop char sits.
84
- * Use for triggers that label short formulaic phrases
85
- * ("State of Delaware") and must not absorb the rest
86
- * of a long forum-selection clause when the comma is
87
- * sentences away. Falls back to the default 100-char
88
- * fallback when omitted.
89
- */
90
- maxLength?: number;
91
+ stop_words?: string[];
92
+ max_length?: number;
91
93
  } | {
92
94
  type: "to-end-of-line";
93
95
  } | {
@@ -97,23 +99,13 @@ type TriggerStrategy = {
97
99
  type: "company-id-value";
98
100
  } | {
99
101
  type: "address";
100
- maxChars?: number;
102
+ max_chars?: number;
101
103
  } | {
102
- /**
103
- * Extract the first regex match in the value text.
104
- * Useful for shape-bounded values that follow a
105
- * label on the same line as other fields, where
106
- * `to-end-of-line` would over-capture. The pattern
107
- * is anchored to the start of the (already
108
- * leading-whitespace-stripped) value, so use
109
- * `(?:.*?)` prefix only when intentional.
110
- */
111
104
  type: "match-pattern";
112
105
  pattern: string;
113
106
  flags?: string;
114
107
  };
115
- /** Validation rules — closed discriminated union. */
116
- type TriggerValidation = {
108
+ type NativeTriggerValidation = {
117
109
  type: "starts-uppercase";
118
110
  } | {
119
111
  type: "min-length";
@@ -129,291 +121,52 @@ type TriggerValidation = {
129
121
  type: "matches-pattern";
130
122
  pattern: string;
131
123
  flags?: string;
132
- }
133
- /**
134
- * Run a named stdnum validator (checksum + length)
135
- * against the captured value. Keeps the trigger
136
- * path symmetrical with the formatted-regex
137
- * detectors so e.g. `CPF nº 00000000000` does not
138
- * survive as a tax-ID entity.
139
- */
140
- | {
141
- type: "valid-id";
142
- validator: ValidIdValidator;
143
- };
144
- /** Built-in stdnum validators that can be referenced
145
- * by `valid-id` validations. */
146
- type ValidIdValidator = "br.cpf" | "br.cnpj" | "us.rtn";
147
- /** Auto-generated trigger variants — closed set. */
148
- type TriggerExtension = "add-colon" | "add-trailing-space" | "add-colon-space" | "normalize-spaces";
149
- /** V2 trigger config entry (JSON shape). */
150
- type TriggerGroupConfig = {
151
- id?: string;
152
- triggers: string[];
153
- label: string;
154
- strategy: TriggerStrategy;
155
- extensions?: TriggerExtension[];
156
- validations?: TriggerValidation[];
157
- /** When true, include the trigger text in the
158
- * entity span (e.g., court names). */
159
- includeTrigger?: boolean;
160
- };
161
- /** Compiled validation with pre-built regex. */
162
- type CompiledValidation = {
163
- type: "starts-uppercase";
164
- re: RegExp;
165
- } | {
166
- type: "min-length";
167
- min: number;
168
- } | {
169
- type: "max-length";
170
- max: number;
171
- } | {
172
- type: "no-digits";
173
- re: RegExp;
174
- } | {
175
- type: "has-digits";
176
- re: RegExp;
177
- } | {
178
- type: "matches-pattern";
179
- re: RegExp;
180
124
  } | {
181
125
  type: "valid-id";
182
- validator: ValidIdValidator;
183
- check: (value: string) => boolean;
126
+ validator: string;
184
127
  };
185
- /**
186
- * Runtime rule — one per trigger string after
187
- * expansion. Fed to the Aho-Corasick automaton.
188
- */
189
- type TriggerRule = {
128
+ type NativeTriggerRule = {
190
129
  trigger: string;
191
130
  label: string;
192
- strategy: TriggerStrategy;
193
- validations: CompiledValidation[];
194
- includeTrigger: boolean;
195
- };
196
- /** Per-label operator selection. Key is the entity label. */
197
- type OperatorConfig = {
198
- /** Operator per label. Missing labels default to "replace". */operators: Record<string, OperatorType>; /** Custom replacement string for the redact operator. */
199
- redactString: string;
131
+ strategy: NativeTriggerStrategy;
132
+ validations: NativeTriggerValidation[];
133
+ include_trigger: boolean;
200
134
  };
201
- /** Whether an operator produces a reversible redaction entry. */
202
- type OperatorReversibility = "reversible" | "irreversible";
203
- type AnonymisationOperator = {
204
- type: OperatorType;
205
- reversibility: OperatorReversibility;
206
- /**
207
- * Apply the operator to a single entity occurrence.
208
- * Returns the replacement string to embed in the document.
209
- */
210
- apply: (text: string, label: string, placeholder: string, redactString: string) => string;
135
+ type NativeTriggerData = {
136
+ rules: NativeTriggerRule[];
137
+ address_stop_keywords: string[];
138
+ party_position_terms: string[];
139
+ post_nominals: string[];
140
+ sentence_terminal_currency_terms: string[];
141
+ phone_extension_labels: string[];
142
+ number_markers: string[];
143
+ number_labels: string[];
211
144
  };
212
- /**
213
- * Redacted document output with stable entity mapping.
214
- */
215
- type RedactionResult = {
216
- redactedText: string;
217
- /**
218
- * Maps placeholder to original text. Only populated for
219
- * reversible operators (replace). Empty for redact.
220
- */
221
- redactionMap: Map<string, string>; /** Maps placeholder to the operator that produced it. */
222
- operatorMap: Map<string, OperatorType>;
223
- entityCount: number;
224
- };
225
- /**
226
- * Configuration for the detection pipeline.
227
- */
228
- type DenyListCategory = "Names" | "Places" | "Addresses" | "Courts" | "Financial" | "Government" | "Healthcare" | "Education" | "Political" | "Organizations" | "International";
229
- /**
230
- * Metadata for a single dictionary entry in the
231
- * deny-list system. Mirrors the shape from
232
- * the anonymize-data package so consumers can pass
233
- * pre-loaded data without a runtime dependency.
234
- */
235
- type DictionaryMeta = {
236
- label: string;
237
- category: DenyListCategory;
238
- country: string | null;
239
- };
240
- /**
241
- * Caller-supplied exact terms for deny-list matching.
242
- * These entries are merged with the published deny-list
243
- * dictionaries when `enableDenyList` is enabled.
244
- */
245
- type CustomDenyListEntry = {
246
- value: string;
247
- label: string;
248
- variants?: readonly string[];
249
- };
250
- /**
251
- * Caller-supplied regex detector. The pattern is passed
252
- * to the underlying text-search regex engine, so use its
253
- * supported regex syntax. Inline flags such as `(?i)` are
254
- * accepted when supported by that engine.
255
- */
256
- type CustomRegexPattern = {
257
- pattern: string;
258
- label: string;
259
- score?: number;
260
- preparedArtifactPolicy?: "include" | "omit";
261
- };
262
- /**
263
- * Pre-loaded dictionary data for dependency injection.
264
- * Consumers that want name/city/deny-list detection
265
- * load dictionaries themselves (e.g. from the
266
- * anonymize-data package) and pass them here; the
267
- * anonymize package has zero cross-package imports.
268
- *
269
- * All fields are optional. When a field is absent,
270
- * the corresponding detection path is skipped (same
271
- * behavior as when no dictionaries are available).
272
- */
273
- type Dictionaries = {
274
- /**
275
- * First names per language code (e.g., "cs", "de").
276
- * Merged with legacy config names at init time.
277
- */
278
- firstNames?: Readonly<Record<string, readonly string[]>>;
279
- /**
280
- * Surnames per language code.
281
- * Merged with legacy config names at init time.
282
- */
283
- surnames?: Readonly<Record<string, readonly string[]>>;
284
- /**
285
- * Non-Western name tokens per locale code
286
- * (e.g., "in", "ar", "ja-latn", "ko", "zh-latn",
287
- * "th", "vi", "fil", "id"). Merged with bundled
288
- * names-nw-*.json data at init time.
289
- */
290
- nonWesternNames?: Readonly<Record<string, readonly string[]>>;
291
- /**
292
- * Pre-loaded deny-list dictionaries keyed by
293
- * dictionary ID (e.g., "courts/CZ", "banks/DE").
294
- * Each value is the array of terms for that
295
- * dictionary.
296
- */
297
- denyList?: Readonly<Record<string, readonly string[]>>;
298
- /**
299
- * Metadata per dictionary ID. Required when
300
- * `denyList` is provided so the pipeline knows
301
- * labels, categories, and country filters.
302
- */
303
- denyListMeta?: Readonly<Record<string, DictionaryMeta>>;
304
- /**
305
- * Pre-loaded city names, already merged across
306
- * all desired countries.
307
- *
308
- * Prefer `citiesByCountry` when callers also pass
309
- * `denyListCountries` / `denyListRegions`; merged
310
- * city arrays cannot be scoped after injection.
311
- */
312
- cities?: readonly string[];
313
- /**
314
- * Pre-loaded city names keyed by ISO 3166-1 alpha-2
315
- * country code. When provided, the deny-list builder
316
- * applies `denyListCountries` / `denyListRegions`
317
- * before adding city patterns to the search automaton.
318
- */
319
- citiesByCountry?: Readonly<Record<string, readonly string[]>>;
145
+ type NativeLegalFormData = {
146
+ suffixes: string[];
147
+ normalized_boundary_suffixes: string[];
148
+ normalized_in_name_words: string[];
149
+ normalized_suffix_words: string[];
150
+ role_heads: string[];
151
+ sentence_verb_indicators: string[];
152
+ clause_noun_heads: string[];
153
+ connector_prose_heads: string[];
154
+ structural_single_cap_prefixes: string[];
155
+ leading_clause_phrases: string[];
156
+ leading_clause_direct_prefixes: string[];
157
+ connector_words: string[];
158
+ and_connector_words: string[];
159
+ in_name_prepositions: string[];
160
+ company_suffix_words: string[];
161
+ comma_gated_direct_prefixes: string[];
320
162
  };
321
- type PipelineConfig = {
322
- threshold: number;
323
- enableTriggerPhrases: boolean;
324
- enableRegex: boolean;
325
- /**
326
- * Expected content language codes. When present, these
327
- * derive default dictionary scopes for name corpus and
328
- * deny-list matching unless the lower-level scope fields
329
- * below are set explicitly.
330
- */
331
- languages?: string[];
332
- /**
333
- * Convenience form for single-language documents. Ignored
334
- * when `languages` is also provided.
335
- */
336
- language?: string;
337
- /**
338
- * Enables legal-form organization detection.
339
- * Required for typed callers; legacy untyped
340
- * callers that omit this field are treated as
341
- * enabled at runtime for backward compatibility.
342
- */
343
- enableLegalForms: boolean;
344
- /**
345
- * Enables first-name/surname/title corpus matching.
346
- * When deny-list mode is enabled, this also controls
347
- * whether name-corpus entries are injected into the
348
- * deny-list search automaton.
349
- */
350
- enableNameCorpus: boolean;
351
- /**
352
- * Optional language scope for first-name/surname
353
- * dictionaries, using the keys present in
354
- * `dictionaries.firstNames` / `dictionaries.surnames`
355
- * (for example `["en", "de"]`). When omitted, all
356
- * injected name languages are used for backward
357
- * compatibility.
358
- */
359
- nameCorpusLanguages?: string[];
360
- enableDenyList: boolean;
361
- denyListCountries?: string[];
362
- denyListRegions?: string[];
363
- denyListExcludeCategories?: string[];
364
- /**
365
- * Caller-owned exact terms to match through the
366
- * deny-list layer. Requires `enableDenyList: true`.
367
- */
368
- customDenyList?: readonly CustomDenyListEntry[];
369
- /**
370
- * Caller-owned regex detectors. Requires
371
- * `enableRegex: true`.
372
- */
373
- customRegexes?: readonly CustomRegexPattern[];
374
- enableGazetteer: boolean;
375
- /**
376
- * Detect country names (ISO 3166-1 names, curated
377
- * aliases, alpha-3 codes). Defaults to true. Names
378
- * span all manifest languages plus widely-used
379
- * additions (Dutch, Russian, Chinese, Arabic, etc.).
380
- */
381
- enableCountries?: boolean;
382
- enableNer: boolean;
383
- enableConfidenceBoost: boolean;
384
- enableCoreference: boolean;
385
- enableZoneClassification?: boolean;
386
- enableHotwordRules?: boolean;
387
- /**
388
- * Requested output labels. An empty array means
389
- * "do not filter by label" for deterministic
390
- * detectors; NER falls back to DEFAULT_ENTITY_LABELS.
391
- */
392
- labels: string[];
393
- workspaceId: string;
394
- /**
395
- * Pre-loaded dictionary data for name, deny-list,
396
- * and city detection. When omitted, dictionary-based
397
- * detection paths are skipped. Consumers load from
398
- * the anonymize-data package and pass the data here.
399
- */
400
- dictionaries?: Dictionaries;
163
+ type NativeDateMonthData = Record<string, string[]>;
164
+ type NativeYearWordData = Record<string, string[]>;
165
+ type NativeDateData = {
166
+ month_names_by_language: NativeDateMonthData;
167
+ year_words_by_language: NativeYearWordData;
401
168
  };
402
- //#endregion
403
- //#region src/detectors/regex.d.ts
404
- type RegexMeta = {
405
- label: string;
406
- score: number;
407
- sourceDetail?: Entity["sourceDetail"];
408
- minByteLength?: number; /** Post-match stdnum validator for confirmation. */
409
- validator?: Validator;
410
- validatorId?: string; /** Extract the identifier portion when context is part of the regex span. */
411
- validatorInput?: (text: string) => string;
412
- validatorInputKind?: "digits-only" | "crypto-wallet-candidate";
413
- };
414
- type DateMonthData = Record<string, string[]>;
415
- type YearWordData = Record<string, string[]>;
416
- type MonetaryData = {
169
+ type NativeMonetaryData = {
417
170
  currencies: {
418
171
  codes: string[];
419
172
  symbols: string[];
@@ -434,337 +187,20 @@ type MonetaryData = {
434
187
  }>;
435
188
  };
436
189
  };
437
- //#endregion
438
- //#region src/context.d.ts
439
- /**
440
- * Compiled RegExp pattern used for coreference
441
- * definition extraction.
442
- */
443
- type DefinitionPattern = {
444
- pattern: RegExp;
445
- };
446
- /**
447
- * Cached data for the name corpus detector.
448
- * Populated by initNameCorpus; consumed by
449
- * detectNameCorpus and deny-list AC integration.
450
- */
451
- type NameCorpusData = {
452
- firstNames: ReadonlySet<string>;
453
- surnames: ReadonlySet<string>;
454
- titleTokens: ReadonlySet<string>;
455
- /** Abbreviation-style titles whose trailing dot is
456
- * part of the title, not a sentence boundary.
457
- * Contains the lowercase, dot-stripped form
458
- * (e.g., "dr", "smt", "atty"). */
459
- titleAbbreviations: ReadonlySet<string>;
460
- excludedWords: ReadonlySet<string>;
461
- /** Lowercased common English words. A name chain whose
462
- * every token is a common word (e.g. "Loan Documents",
463
- * where "Loan" coincides with a Vietnamese given name)
464
- * is treated as a common-word phrase, not a person. */
465
- commonWords: ReadonlySet<string>; /** Non-Western name tokens merged across all locales. */
466
- nonWesternNames: ReadonlySet<string>; /** All-caps acronyms excluded from name detection. */
467
- excludedAllCaps: ReadonlySet<string>; /** Raw arrays exposed for deny-list AC integration. */
468
- firstNamesList: readonly string[];
469
- surnamesList: readonly string[];
470
- titlesList: readonly string[];
471
- excludedList: readonly string[];
472
- nonWesternNamesList: readonly string[];
473
- excludedAllCapsList: readonly string[];
474
- };
475
- /**
476
- * All cached state for a single pipeline run (or
477
- * sequence of runs sharing the same config). Replacing
478
- * module-level singletons with this object enables
479
- * concurrent pipelines with different configs and
480
- * simplifies testing.
481
- *
482
- * Each field starts null and is populated lazily on
483
- * first use by the corresponding loader function.
484
- */
485
- type PipelineContext = {
486
- search: UnifiedSearchInstance | null;
487
- searchKey: string;
488
- searchPromise: Promise<UnifiedSearchInstance> | null;
489
- nativePipelinePackage: Uint8Array | null;
490
- nativePipelinePackageKey: string;
491
- nativePipelinePackagePromise: Promise<Uint8Array> | null;
492
- nameCorpus: NameCorpusData | null;
493
- nameCorpusKey: string;
494
- nameCorpusPromise: Promise<void> | null;
495
- stopwords: ReadonlySet<string> | null;
496
- stopwordsPromise: Promise<ReadonlySet<string>> | null;
497
- allowList: ReadonlySet<string> | null;
498
- allowListPromise: Promise<ReadonlySet<string>> | null;
499
- personStopwords: ReadonlySet<string> | null;
500
- personStopwordsPromise: Promise<ReadonlySet<string>> | null;
501
- definedTermHeads: ReadonlySet<string> | null;
502
- definedTermHeadsPromise: Promise<ReadonlySet<string>> | null;
503
- addressStopwords: ReadonlySet<string> | null;
504
- addressStopwordsPromise: Promise<ReadonlySet<string>> | null; /** First-name exclusions for stopword filtering. */
505
- firstNameExclusions: ReadonlySet<string> | null;
506
- firstNameExclusionCorpusLen: number;
507
- genericRoles: ReadonlySet<string> | null;
508
- genericRolesPromise: Promise<ReadonlySet<string>> | null;
509
- corefPatterns: DefinitionPattern[] | null;
510
- corefPatternsKey: string;
511
- corefPatternsPromise: Promise<DefinitionPattern[]> | null;
512
- corefLoadAttempted: boolean;
513
- roleStopSet: ReadonlySet<string> | null;
514
- roleStopSetPromise: Promise<ReadonlySet<string>> | null;
515
- zoneHeadingPatterns: RegExp[] | null;
516
- zoneSigningPatterns: RegExp[] | null;
517
- zoneInitPromise: Promise<void> | null;
518
- };
519
- //#endregion
520
- //#region src/detectors/deny-list.d.ts
521
- type DenyListFilterData = {
522
- stopwords: string[];
523
- allowList: string[];
524
- personStopwords: string[];
525
- personTrailingNouns: string[];
526
- addressStopwords: string[];
527
- addressJurisdictionPrefixes: string[];
528
- streetTypes: string[];
529
- addressComponentTerms: string[];
530
- ambiguousStreetTypeTerms: string[];
531
- firstNames: string[];
532
- genericRoles: string[];
533
- numberAbbrevPrefixes: string[];
534
- sentenceStarters: string[];
535
- trailingAddressWordExclusions: string[];
536
- documentHeadingWords: string[];
537
- documentHeadingOrdinalMarkers: string[];
538
- definedTermCues: string[];
539
- signingPlaceGuards: DenyListSigningPlaceGuardData[];
540
- };
541
- type DenyListSigningPlaceGuardData = {
542
- prefixPhrases: string[];
543
- suffixPhrases: string[];
544
- };
545
- /**
546
- * Source tag for each pattern in the automaton.
547
- * "deny-list" = standard deny list entry
548
- * "city" = city dictionary entry
549
- * "custom-deny-list" = caller-owned exact term
550
- * "first-name" = name corpus first name
551
- * "surname" = name corpus surname
552
- * "title" = academic/professional title
553
- */
554
- type PatternSource = "deny-list" | "city" | "custom-deny-list" | "first-name" | "surname" | "title";
555
- type PatternLabels = string | string[];
556
- type PatternSources = PatternSource | PatternSource[];
557
- /**
558
- * Pre-built deny list data. Constructed once by
559
- * `buildDenyList`, reused across `processDenyListMatches`
560
- * calls. Contains PatternEntry[] for the unified builder
561
- * plus parallel label/source arrays for post-processing.
562
- */
563
- type DenyListData = {
564
- /**
565
- * Maps pattern index → entity labels (plural).
566
- * Same pattern can have multiple labels when it
567
- * appears in multiple dictionaries (e.g., "Denver"
568
- * is both a person name and a city name).
569
- */
570
- labels: PatternLabels[]; /** Maps pattern index → labels contributed by custom entries. */
571
- customLabels: (PatternLabels | undefined)[]; /** Maps pattern index → original pattern text. */
572
- originals: string[]; /** Maps pattern index → source types (plural). */
573
- sources: PatternSources[];
574
- filters: DenyListFilterData;
575
- };
576
- //#endregion
577
- //#region src/detectors/address-seeds.d.ts
578
- type AddressSeedData = {
190
+ type NativeAddressSeedData = {
579
191
  boundary_words: string[];
580
192
  br_cep_cue_words: string[];
581
193
  unit_abbreviations: string[];
582
194
  };
583
- //#endregion
584
- //#region src/detectors/countries.d.ts
585
- /**
586
- * Pre-built country patterns + parallel label/source
587
- * metadata. Constructed once and reused across pipeline
588
- * runs.
589
- */
590
- type CountryData = {
591
- /** Maps local pattern index to entity label. Always "country". */labels: string[];
592
- /**
593
- * Maps local pattern index to the alpha-2 ISO code the
594
- * pattern resolves to. Used for downstream coreference /
595
- * placeholder grouping.
596
- */
597
- isoCodes: string[]; /** Maps local pattern index to pattern variant kind. */
598
- variants: CountryVariant[];
599
- };
600
- type CountryVariant = "name" | "alias" | "alpha3" | "alpha2";
601
- //#endregion
602
- //#region src/filters/confidence-boost.d.ts
603
- type AddressContextData = {
195
+ type NativeAddressContextData = {
604
196
  address_prepositions: string[];
605
197
  temporal_prepositions: string[];
606
198
  street_abbreviations: string[];
607
199
  bare_house_stopwords: string[];
608
200
  };
609
- //#endregion
610
- //#region src/build-unified-search.d.ts
611
- type PatternSlice = {
612
- start: number;
613
- end: number;
614
- };
615
- type NativeSearchPatternKind = "literal" | "literal-with-options" | "regex" | "fuzzy";
616
- type NativeSearchPattern = {
617
- kind: NativeSearchPatternKind;
201
+ type NativeCoreferencePatternData = {
618
202
  pattern: string;
619
- distance?: number;
620
- case_insensitive?: boolean;
621
- whole_words?: boolean;
622
- lazy?: boolean;
623
- prefilter_any?: string[];
624
- prefilter_case_insensitive?: boolean;
625
- prefilter_regex?: string;
626
- prefilter_window_bytes?: number;
627
- prepared_artifact_policy?: "include" | "omit";
628
- };
629
- type NativeSearchOptions = {
630
- literal_case_insensitive?: boolean;
631
- literal_whole_words?: boolean;
632
- regex_whole_words?: boolean;
633
- regex_overlap_all?: boolean;
634
- regex_artifact_policy?: "include" | "omit";
635
- fuzzy_case_insensitive?: boolean;
636
- fuzzy_whole_words?: boolean;
637
- fuzzy_normalize_diacritics?: boolean;
638
- };
639
- type NativeRegexMatchMeta = {
640
- label: string;
641
- score: number;
642
- source_detail?: string;
643
- requires_validation?: boolean;
644
- validator_id?: string;
645
- validator_input?: string;
646
- min_byte_length?: number;
647
- };
648
- type NativeDenyListFilterData = {
649
- stopwords: string[];
650
- allow_list: string[];
651
- person_stopwords: string[];
652
- person_trailing_nouns: string[];
653
- address_stopwords: string[];
654
- address_jurisdiction_prefixes: string[];
655
- street_types: string[];
656
- address_component_terms: string[];
657
- ambiguous_street_type_terms: string[];
658
- first_names: string[];
659
- generic_roles: string[];
660
- number_abbrev_prefixes: string[];
661
- sentence_starters: string[];
662
- trailing_address_word_exclusions: string[];
663
- document_heading_words: string[];
664
- document_heading_ordinal_markers: string[];
665
- defined_term_cues: string[];
666
- signing_place_guards: NativeSigningPlaceGuardData[];
667
- };
668
- type NativeSigningPlaceGuardData = {
669
- prefix_phrases: string[];
670
- suffix_phrases: string[];
671
- };
672
- type NativeDenyListMatchData = {
673
- labels?: string[][];
674
- label_table?: string[];
675
- label_indices?: number[][];
676
- custom_labels?: string[][];
677
- custom_label_indices?: number[][];
678
- originals: string[];
679
- sources?: string[][];
680
- source_table?: string[];
681
- source_indices?: number[][];
682
- filters?: NativeDenyListFilterData;
683
- };
684
- type NativeTriggerStrategy = {
685
- type: "to-next-comma";
686
- stop_words?: string[];
687
- max_length?: number;
688
- } | {
689
- type: "to-end-of-line";
690
- } | {
691
- type: "n-words";
692
- count: number;
693
- } | {
694
- type: "company-id-value";
695
- } | {
696
- type: "address";
697
- max_chars?: number;
698
- } | {
699
- type: "match-pattern";
700
- pattern: string;
701
- flags?: string;
702
- };
703
- type NativeTriggerValidation = {
704
- type: "starts-uppercase";
705
- } | {
706
- type: "min-length";
707
- min: number;
708
- } | {
709
- type: "max-length";
710
- max: number;
711
- } | {
712
- type: "no-digits";
713
- } | {
714
- type: "has-digits";
715
- } | {
716
- type: "matches-pattern";
717
- pattern: string;
718
- flags?: string;
719
- } | {
720
- type: "valid-id";
721
- validator: string;
722
- };
723
- type NativeTriggerRule = {
724
- trigger: string;
725
- label: string;
726
- strategy: NativeTriggerStrategy;
727
- validations: NativeTriggerValidation[];
728
- include_trigger: boolean;
729
- };
730
- type NativeTriggerData = {
731
- rules: NativeTriggerRule[];
732
- address_stop_keywords: string[];
733
- party_position_terms: string[];
734
- post_nominals: string[];
735
- sentence_terminal_currency_terms: string[];
736
- phone_extension_labels: string[];
737
- number_markers: string[];
738
- number_labels: string[];
739
- };
740
- type NativeLegalFormData = {
741
- suffixes: string[];
742
- normalized_boundary_suffixes: string[];
743
- normalized_in_name_words: string[];
744
- normalized_suffix_words: string[];
745
- role_heads: string[];
746
- sentence_verb_indicators: string[];
747
- clause_noun_heads: string[];
748
- connector_prose_heads: string[];
749
- structural_single_cap_prefixes: string[];
750
- leading_clause_phrases: string[];
751
- leading_clause_direct_prefixes: string[];
752
- connector_words: string[];
753
- and_connector_words: string[];
754
- in_name_prepositions: string[];
755
- company_suffix_words: string[];
756
- comma_gated_direct_prefixes: string[];
757
- };
758
- type NativeDateData = {
759
- month_names_by_language: DateMonthData;
760
- year_words_by_language: YearWordData;
761
- };
762
- type NativeMonetaryData = MonetaryData;
763
- type NativeAddressSeedData = AddressSeedData;
764
- type NativeAddressContextData = AddressContextData;
765
- type NativeCoreferencePatternData = {
766
- pattern: string;
767
- flags: string;
203
+ flags: string;
768
204
  };
769
205
  type NativeCoreferenceData = {
770
206
  definition_patterns: NativeCoreferencePatternData[];
@@ -804,6 +240,11 @@ type NativeZoneData = {
804
240
  section_heading_patterns: NativeZonePatternData[];
805
241
  signing_clauses: NativeZoneSigningClauseData[];
806
242
  };
243
+ type NativeCountryData = {
244
+ labels: string[];
245
+ isoCodes: string[];
246
+ variants: Array<"name" | "alias" | "alpha3" | "alpha2">;
247
+ };
807
248
  type NativeGazetteerData = {
808
249
  labels: string[];
809
250
  is_fuzzy: boolean[];
@@ -855,7 +296,7 @@ type NativePreparedSearchConfig = {
855
296
  deny_list_data?: NativeDenyListMatchData;
856
297
  false_positive_filters?: NativeDenyListFilterData;
857
298
  gazetteer_data?: NativeGazetteerData;
858
- country_data?: CountryData;
299
+ country_data?: NativeCountryData;
859
300
  hotword_data?: NativeHotwordRuleData;
860
301
  trigger_data?: NativeTriggerData;
861
302
  legal_form_data?: NativeLegalFormData;
@@ -869,35 +310,403 @@ type NativePreparedSearchConfig = {
869
310
  date_data?: NativeDateData;
870
311
  monetary_data?: NativeMonetaryData;
871
312
  };
872
- type GazetteerData = {
873
- /** Maps local pattern index to entry label. */labels: string[];
313
+ //#endregion
314
+ //#region src/types.d.ts
315
+ /**
316
+ * Fields shared by every entity span in the source text.
317
+ */
318
+ type EntityBase = {
319
+ start: number;
320
+ end: number;
321
+ label: string;
322
+ text: string;
323
+ score: number;
324
+ sourceDetail?: "custom-deny-list" | "custom-regex" | "gazetteer-extension";
325
+ };
326
+ /**
327
+ * A PII entity span found by a primary detection layer
328
+ * (regex, NER, legal forms, deny list, ...).
329
+ */
330
+ type DetectedEntity = EntityBase & {
331
+ source: Exclude<DetectionSource, typeof DETECTION_SOURCES.COREFERENCE>;
332
+ };
333
+ /**
334
+ * An alias mention of a previously detected entity: a
335
+ * defined term ("the Seller") or a propagated bare
336
+ * mention ("Acme" after "Acme Corp.").
337
+ *
338
+ * `corefSourceText` is required by construction, so an
339
+ * alias cannot exist without the link back to its source
340
+ * entity. Placeholder numbering reads it to give the
341
+ * alias the same placeholder as the source. The link
342
+ * travels with the entity instead of living in a
343
+ * side-channel map that a producer could forget to
344
+ * write — or that a later pass could clear.
345
+ */
346
+ type CorefAliasEntity = EntityBase & {
347
+ source: typeof DETECTION_SOURCES.COREFERENCE; /** Full text of the source entity this alias refers to. */
348
+ corefSourceText: string;
349
+ };
350
+ /**
351
+ * A detected PII entity span in the source text.
352
+ * Every detection layer produces these.
353
+ */
354
+ type Entity = DetectedEntity | CorefAliasEntity;
355
+ /**
356
+ * Entity after human review. Extends the base Entity
357
+ * with a review decision.
358
+ */
359
+ type ReviewDecision = "confirmed" | "rejected" | "relabeled";
360
+ type ReviewedEntity = Entity & {
361
+ decision?: ReviewDecision;
362
+ originalLabel?: string;
363
+ };
364
+ /**
365
+ * A single entry in the workspace-scoped gazetteer
366
+ * (deny list). Persisted in IndexedDB.
367
+ */
368
+ type GazetteerEntry = {
369
+ id: string;
370
+ canonical: string;
371
+ label: string;
372
+ variants: string[];
373
+ workspaceId: string;
374
+ createdAt: number;
375
+ source: "manual" | "confirmed-from-model";
376
+ };
377
+ /** Extraction strategy — closed discriminated union. */
378
+ type TriggerStrategy = {
379
+ type: "to-next-comma";
380
+ /**
381
+ * Optional list of lowercase keywords that terminate
382
+ * the value scan, in addition to commas/newlines. Useful
383
+ * for triggers like court names that may continue past
384
+ * a missing comma into adjacent clause text ("Městským
385
+ * soudem v Praze dne 1. 1. 2020"); listing `"dne"` here
386
+ * stops the scan at the date boundary. Matched on a
387
+ * word-boundary, case-insensitive.
388
+ */
389
+ stopWords?: string[];
390
+ /**
391
+ * Hard cap on the captured span length, in characters,
392
+ * regardless of where the next comma / stop char sits.
393
+ * Use for triggers that label short formulaic phrases
394
+ * ("State of Delaware") and must not absorb the rest
395
+ * of a long forum-selection clause when the comma is
396
+ * sentences away. Falls back to the default 100-char
397
+ * fallback when omitted.
398
+ */
399
+ maxLength?: number;
400
+ } | {
401
+ type: "to-end-of-line";
402
+ } | {
403
+ type: "n-words";
404
+ count: number;
405
+ } | {
406
+ type: "company-id-value";
407
+ } | {
408
+ type: "address";
409
+ maxChars?: number;
410
+ } | {
874
411
  /**
875
- * Whether each pattern is fuzzy (distance > 0).
876
- * Used by the post-processor to assign scores.
412
+ * Extract the first regex match in the value text.
413
+ * Useful for shape-bounded values that follow a
414
+ * label on the same line as other fields, where
415
+ * `to-end-of-line` would over-capture. The pattern
416
+ * is anchored to the start of the (already
417
+ * leading-whitespace-stripped) value, so use
418
+ * `(?:.*?)` prefix only when intentional.
877
419
  */
878
- isFuzzy: boolean[];
420
+ type: "match-pattern";
421
+ pattern: string;
422
+ flags?: string;
879
423
  };
880
- type UnifiedSearchInstance = {
881
- /** Regex + triggers + legal-forms. */tsRegex: TextSearch; /** Caller-owned custom regexes, isolated for overlap preservation. */
882
- tsCustomRegex: TextSearch; /** Deny-list + street-types + gazetteer. */
883
- tsLiterals: TextSearch;
884
- slices: {
885
- regex: PatternSlice;
886
- customRegex: PatternSlice;
887
- legalForms: PatternSlice;
888
- triggers: PatternSlice;
889
- denyList: PatternSlice;
890
- streetTypes: PatternSlice;
891
- gazetteer: PatternSlice;
892
- countries: PatternSlice;
893
- };
894
- regexMeta: readonly RegexMeta[];
895
- customRegexMeta: readonly RegexMeta[];
896
- triggerRules: readonly TriggerRule[];
897
- denyListData: DenyListData | null;
898
- gazetteerData: GazetteerData | null;
899
- countryData: CountryData | null;
900
- nativeStaticConfig: NativePreparedSearchConfig;
424
+ /** Validation rules — closed discriminated union. */
425
+ type TriggerValidation = {
426
+ type: "starts-uppercase";
427
+ } | {
428
+ type: "min-length";
429
+ min: number;
430
+ } | {
431
+ type: "max-length";
432
+ max: number;
433
+ } | {
434
+ type: "no-digits";
435
+ } | {
436
+ type: "has-digits";
437
+ } | {
438
+ type: "matches-pattern";
439
+ pattern: string;
440
+ flags?: string;
441
+ }
442
+ /**
443
+ * Run a named stdnum validator (checksum + length)
444
+ * against the captured value. Keeps the trigger
445
+ * path symmetrical with the formatted-regex
446
+ * detectors so e.g. `CPF nº 00000000000` does not
447
+ * survive as a tax-ID entity.
448
+ */
449
+ | {
450
+ type: "valid-id";
451
+ validator: ValidIdValidator;
452
+ };
453
+ /** Built-in stdnum validators that can be referenced
454
+ * by `valid-id` validations. */
455
+ type ValidIdValidator = "br.cpf" | "br.cnpj" | "us.rtn";
456
+ /** Auto-generated trigger variants — closed set. */
457
+ type TriggerExtension = "add-colon" | "add-trailing-space" | "add-colon-space" | "normalize-spaces";
458
+ /** V2 trigger config entry (JSON shape). */
459
+ type TriggerGroupConfig = {
460
+ id?: string;
461
+ triggers: string[];
462
+ label: string;
463
+ strategy: TriggerStrategy;
464
+ extensions?: TriggerExtension[];
465
+ validations?: TriggerValidation[];
466
+ /** When true, include the trigger text in the
467
+ * entity span (e.g., court names). */
468
+ includeTrigger?: boolean;
469
+ };
470
+ /** Compiled validation with pre-built regex. */
471
+ type CompiledValidation = {
472
+ type: "starts-uppercase";
473
+ re: RegExp;
474
+ } | {
475
+ type: "min-length";
476
+ min: number;
477
+ } | {
478
+ type: "max-length";
479
+ max: number;
480
+ } | {
481
+ type: "no-digits";
482
+ re: RegExp;
483
+ } | {
484
+ type: "has-digits";
485
+ re: RegExp;
486
+ } | {
487
+ type: "matches-pattern";
488
+ re: RegExp;
489
+ } | {
490
+ type: "valid-id";
491
+ validator: ValidIdValidator;
492
+ check: (value: string) => boolean;
493
+ };
494
+ /**
495
+ * Runtime rule — one per trigger string after
496
+ * expansion. Fed to the Aho-Corasick automaton.
497
+ */
498
+ type TriggerRule = {
499
+ trigger: string;
500
+ label: string;
501
+ strategy: TriggerStrategy;
502
+ validations: CompiledValidation[];
503
+ includeTrigger: boolean;
504
+ };
505
+ /** Per-label operator selection. Key is the entity label. */
506
+ type OperatorConfig = {
507
+ /** Operator per label. Missing labels default to "replace". */operators: Record<string, OperatorType>; /** Custom replacement string for the redact operator. */
508
+ redactString: string;
509
+ };
510
+ /** Whether an operator produces a reversible redaction entry. */
511
+ type OperatorReversibility = "reversible" | "irreversible";
512
+ type AnonymisationOperator = {
513
+ type: OperatorType;
514
+ reversibility: OperatorReversibility;
515
+ /**
516
+ * Apply the operator to a single entity occurrence.
517
+ * Returns the replacement string to embed in the document.
518
+ */
519
+ apply: (text: string, label: string, placeholder: string, redactString: string) => string;
520
+ };
521
+ /**
522
+ * Redacted document output with stable entity mapping.
523
+ */
524
+ type RedactionResult = {
525
+ redactedText: string;
526
+ /**
527
+ * Maps placeholder to original text. Only populated for
528
+ * reversible operators (replace). Empty for redact.
529
+ */
530
+ redactionMap: Map<string, string>; /** Maps placeholder to the operator that produced it. */
531
+ operatorMap: Map<string, OperatorType>;
532
+ entityCount: number;
533
+ };
534
+ /**
535
+ * Configuration for the detection pipeline.
536
+ */
537
+ type DenyListCategory = "Names" | "Places" | "Addresses" | "Courts" | "Financial" | "Government" | "Healthcare" | "Education" | "Political" | "Organizations" | "International";
538
+ /**
539
+ * Metadata for a single dictionary entry in the
540
+ * deny-list system. Mirrors the shape from
541
+ * the anonymize-data package so consumers can pass
542
+ * pre-loaded data without a runtime dependency.
543
+ */
544
+ type DictionaryMeta = {
545
+ label: string;
546
+ category: DenyListCategory;
547
+ country: string | null;
548
+ };
549
+ /**
550
+ * Caller-supplied exact terms for deny-list matching.
551
+ * These entries are merged with the published deny-list
552
+ * dictionaries when `enableDenyList` is enabled.
553
+ */
554
+ type CustomDenyListEntry = {
555
+ value: string;
556
+ label: string;
557
+ variants?: readonly string[];
558
+ };
559
+ /**
560
+ * Caller-supplied regex detector. The pattern is passed
561
+ * to the underlying text-search regex engine, so use its
562
+ * supported regex syntax. Inline flags such as `(?i)` are
563
+ * accepted when supported by that engine.
564
+ */
565
+ type CustomRegexPattern = {
566
+ pattern: string;
567
+ label: string;
568
+ score?: number;
569
+ preparedArtifactPolicy?: "include" | "omit";
570
+ };
571
+ /**
572
+ * Pre-loaded dictionary data for dependency injection.
573
+ * Consumers that want name/city/deny-list detection
574
+ * load dictionaries themselves (e.g. from the
575
+ * anonymize-data package) and pass them here; the
576
+ * anonymize package has zero cross-package imports.
577
+ *
578
+ * All fields are optional. When a field is absent,
579
+ * the corresponding detection path is skipped (same
580
+ * behavior as when no dictionaries are available).
581
+ */
582
+ type Dictionaries = {
583
+ /**
584
+ * First names per language code (e.g., "cs", "de").
585
+ * Merged with legacy config names at init time.
586
+ */
587
+ firstNames?: Readonly<Record<string, readonly string[]>>;
588
+ /**
589
+ * Surnames per language code.
590
+ * Merged with legacy config names at init time.
591
+ */
592
+ surnames?: Readonly<Record<string, readonly string[]>>;
593
+ /**
594
+ * Non-Western name tokens per locale code
595
+ * (e.g., "in", "ar", "ja-latn", "ko", "zh-latn",
596
+ * "th", "vi", "fil", "id"). Merged with bundled
597
+ * names-nw-*.json data at init time.
598
+ */
599
+ nonWesternNames?: Readonly<Record<string, readonly string[]>>;
600
+ /**
601
+ * Pre-loaded deny-list dictionaries keyed by
602
+ * dictionary ID (e.g., "courts/CZ", "banks/DE").
603
+ * Each value is the array of terms for that
604
+ * dictionary.
605
+ */
606
+ denyList?: Readonly<Record<string, readonly string[]>>;
607
+ /**
608
+ * Metadata per dictionary ID. Required when
609
+ * `denyList` is provided so the pipeline knows
610
+ * labels, categories, and country filters.
611
+ */
612
+ denyListMeta?: Readonly<Record<string, DictionaryMeta>>;
613
+ /**
614
+ * Pre-loaded city names, already merged across
615
+ * all desired countries.
616
+ *
617
+ * Prefer `citiesByCountry` when callers also pass
618
+ * `denyListCountries` / `denyListRegions`; merged
619
+ * city arrays cannot be scoped after injection.
620
+ */
621
+ cities?: readonly string[];
622
+ /**
623
+ * Pre-loaded city names keyed by ISO 3166-1 alpha-2
624
+ * country code. When provided, the deny-list builder
625
+ * applies `denyListCountries` / `denyListRegions`
626
+ * before adding city patterns to the search automaton.
627
+ */
628
+ citiesByCountry?: Readonly<Record<string, readonly string[]>>;
629
+ };
630
+ type PipelineConfig = {
631
+ threshold: number;
632
+ enableTriggerPhrases: boolean;
633
+ enableRegex: boolean;
634
+ /**
635
+ * Expected content language codes. When present, these
636
+ * derive default dictionary scopes for name corpus and
637
+ * deny-list matching unless the lower-level scope fields
638
+ * below are set explicitly.
639
+ */
640
+ languages?: string[];
641
+ /**
642
+ * Convenience form for single-language documents. Ignored
643
+ * when `languages` is also provided.
644
+ */
645
+ language?: string;
646
+ /**
647
+ * Enables legal-form organization detection.
648
+ * Required for typed callers; legacy untyped
649
+ * callers that omit this field are treated as
650
+ * enabled at runtime for backward compatibility.
651
+ */
652
+ enableLegalForms: boolean;
653
+ /**
654
+ * Enables first-name/surname/title corpus matching.
655
+ * When deny-list mode is enabled, this also controls
656
+ * whether name-corpus entries are injected into the
657
+ * deny-list search automaton.
658
+ */
659
+ enableNameCorpus: boolean;
660
+ /**
661
+ * Optional language scope for first-name/surname
662
+ * dictionaries, using the keys present in
663
+ * `dictionaries.firstNames` / `dictionaries.surnames`
664
+ * (for example `["en", "de"]`). When omitted, all
665
+ * injected name languages are used for backward
666
+ * compatibility.
667
+ */
668
+ nameCorpusLanguages?: string[];
669
+ enableDenyList: boolean;
670
+ denyListCountries?: string[];
671
+ denyListRegions?: string[];
672
+ denyListExcludeCategories?: string[];
673
+ /**
674
+ * Caller-owned exact terms to match through the
675
+ * deny-list layer. Requires `enableDenyList: true`.
676
+ */
677
+ customDenyList?: readonly CustomDenyListEntry[];
678
+ /**
679
+ * Caller-owned regex detectors. Requires
680
+ * `enableRegex: true`.
681
+ */
682
+ customRegexes?: readonly CustomRegexPattern[];
683
+ enableGazetteer: boolean;
684
+ /**
685
+ * Detect country names (ISO 3166-1 names, curated
686
+ * aliases, alpha-3 codes). Defaults to true. Names
687
+ * span all manifest languages plus widely-used
688
+ * additions (Dutch, Russian, Chinese, Arabic, etc.).
689
+ */
690
+ enableCountries?: boolean;
691
+ enableNer: boolean;
692
+ enableConfidenceBoost: boolean;
693
+ enableCoreference: boolean;
694
+ enableZoneClassification?: boolean;
695
+ enableHotwordRules?: boolean;
696
+ /**
697
+ * Requested output labels. An empty array means
698
+ * "do not filter by label" for deterministic
699
+ * detectors; NER falls back to DEFAULT_ENTITY_LABELS.
700
+ */
701
+ labels: string[];
702
+ workspaceId: string;
703
+ /**
704
+ * Pre-loaded dictionary data for name, deny-list,
705
+ * and city detection. When omitted, dictionary-based
706
+ * detection paths are skipped. Consumers load from
707
+ * the anonymize-data package and pass the data here.
708
+ */
709
+ dictionaries?: Dictionaries;
901
710
  };
902
711
  //#endregion
903
712
  //#region src/native.d.ts
@@ -959,6 +768,9 @@ type NativeAnonymizeBinding = {
959
768
  };
960
769
  prepareStaticSearchPackageBytes: (configJson: Uint8Array) => Uint8Array;
961
770
  prepareStaticSearchCompressedPackageBytes: (configJson: Uint8Array) => Uint8Array;
771
+ assembleStaticSearchConfigJson?: (pipelineConfigJson: Uint8Array, dictionariesJson?: Uint8Array, gazetteerJson?: Uint8Array) => Uint8Array;
772
+ assembleStaticSearchPackageBytes?: (pipelineConfigJson: Uint8Array, dictionariesJson?: Uint8Array, gazetteerJson?: Uint8Array) => Uint8Array;
773
+ assembleStaticSearchCompressedPackageBytes?: (pipelineConfigJson: Uint8Array, dictionariesJson?: Uint8Array, gazetteerJson?: Uint8Array) => Uint8Array;
962
774
  };
963
775
  type NativeOperatorConfig = {
964
776
  operators?: Record<string, OperatorType>;
@@ -1154,5 +966,5 @@ type PreparedSearch = PreparedNativeAnonymizer;
1154
966
  declare const PreparedAnonymizer: typeof PreparedNativeAnonymizer;
1155
967
  type PreparedAnonymizer = PreparedNativeAnonymizer;
1156
968
  //#endregion
1157
- export { Entity as $, createNativePipelineFromPackage as A, prepare_search_package as B, SharedNativeRedactTextJsonOptions as C, assertNativeBindingVersion as D, SharedNativeSearchPackageOptions as E, getNativeBindingVersion as F, NativePreparedSearchConfig as G, redact_text_json as H, load_prepared_package as I, CustomDenyListEntry as J, PipelineContext as K, native_package_version as L, diagnostics_stream_json as M, encodeNativeSearchConfig as N, createNativeAnonymizerFromConfig as O, encodeNativeSearchConfigInput as P, DictionaryMeta as Q, normalize_for_search as R, SharedNativePreparedPackageOptions as S, SharedNativeRedactTextStreamJsonOptions as T, redact_text_stream_json as U, redact_text as V, summary_diagnostics_json as W, DenyListCategory as X, CustomRegexPattern as Y, Dictionaries as Z, PreparedNativeAnonymizer as _, NativeDiagnosticsBatchCallback as a, ReviewedEntity as at, SharedNativeDiagnosticsJsonOptions as b, NativePipelineEntity as c, TriggerRule as ct, NativeRedactionResult as d, GazetteerEntry as et, NativeResultEventCallback as f, PreparedAnonymizer as g, NativeStaticRedactionResult as h, NativeBindingVersionOptions as i, ReviewDecision as it, diagnostics_json as j, createNativeAnonymizerFromPackage as k, NativePipelineFromPackageOptions as l, TriggerStrategy as lt, NativeSearchPackageOptions as m, NativeAnonymizerFromConfigOptions as n, PipelineConfig as nt, NativeNormalizeOptions as o, TriggerExtension as ot, NativeSearchPackageInput as p, AnonymisationOperator as q, NativeAnonymizerFromPackageOptions as r, RedactionResult as rt, NativeOperatorConfig as s, TriggerGroupConfig as st, NativeAnonymizeBinding as t, OperatorConfig as tt, NativePreparedSearchBinding as u, TriggerValidation as ut, PreparedNativePipeline as v, SharedNativeRedactTextOptions as w, SharedNativeDiagnosticsStreamJsonOptions as x, PreparedSearch as y, prepareNativeSearchPackage as z };
969
+ export { OperatorConfig as $, createNativePipelineFromPackage as A, prepare_search_package as B, SharedNativeRedactTextJsonOptions as C, assertNativeBindingVersion as D, SharedNativeSearchPackageOptions as E, getNativeBindingVersion as F, AnonymisationOperator as G, redact_text_json as H, load_prepared_package as I, DenyListCategory as J, CustomDenyListEntry as K, native_package_version as L, diagnostics_stream_json as M, encodeNativeSearchConfig as N, createNativeAnonymizerFromConfig as O, encodeNativeSearchConfigInput as P, GazetteerEntry as Q, normalize_for_search as R, SharedNativePreparedPackageOptions as S, SharedNativeRedactTextStreamJsonOptions as T, redact_text_stream_json as U, redact_text as V, summary_diagnostics_json as W, DictionaryMeta as X, Dictionaries as Y, Entity as Z, PreparedNativeAnonymizer as _, NativeDiagnosticsBatchCallback as a, TriggerGroupConfig as at, SharedNativeDiagnosticsJsonOptions as b, NativePipelineEntity as c, TriggerValidation as ct, NativeRedactionResult as d, PipelineConfig as et, NativeResultEventCallback as f, PreparedAnonymizer as g, NativeStaticRedactionResult as h, NativeBindingVersionOptions as i, TriggerExtension as it, diagnostics_json as j, createNativeAnonymizerFromPackage as k, NativePipelineFromPackageOptions as l, NativePreparedSearchConfig as lt, NativeSearchPackageOptions as m, NativeAnonymizerFromConfigOptions as n, ReviewDecision as nt, NativeNormalizeOptions as o, TriggerRule as ot, NativeSearchPackageInput as p, CustomRegexPattern as q, NativeAnonymizerFromPackageOptions as r, ReviewedEntity as rt, NativeOperatorConfig as s, TriggerStrategy as st, NativeAnonymizeBinding as t, RedactionResult as tt, NativePreparedSearchBinding as u, PreparedNativePipeline as v, SharedNativeRedactTextOptions as w, SharedNativeDiagnosticsStreamJsonOptions as x, PreparedSearch as y, prepareNativeSearchPackage as z };
1158
970
  //# sourceMappingURL=native.d.mts.map