@stll/anonymize 1.5.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (155) hide show
  1. package/ATTRIBUTION.md +70 -0
  2. package/README.md +86 -48
  3. package/dist/index.d.mts +3 -1201
  4. package/dist/index.mjs +3 -16267
  5. package/dist/index.mjs.map +1 -1
  6. package/dist/native-node.d.mts +138 -0
  7. package/dist/native-node.mjs +3 -0
  8. package/dist/native-node2.d.mts +3 -0
  9. package/dist/native-node2.mjs +728 -0
  10. package/dist/native-node2.mjs.map +1 -0
  11. package/dist/native.d.mts +970 -0
  12. package/dist/native.mjs +230 -0
  13. package/dist/native.mjs.map +1 -0
  14. package/dist/native2.d.mts +2 -0
  15. package/index.cjs +3 -0
  16. package/native-pipeline.cs.stlanonpkg +0 -0
  17. package/native-pipeline.de.stlanonpkg +0 -0
  18. package/native-pipeline.en.stlanonpkg +0 -0
  19. package/native-pipeline.stlanonpkg +0 -0
  20. package/package.json +57 -9
  21. package/scripts/build-native-pipeline-package.mjs +225 -0
  22. package/dist/address-boundaries.mjs +0 -195
  23. package/dist/address-boundaries.mjs.map +0 -1
  24. package/dist/address-prepositions.mjs +0 -182
  25. package/dist/address-prepositions.mjs.map +0 -1
  26. package/dist/address-stop-keywords.mjs +0 -137
  27. package/dist/address-stop-keywords.mjs.map +0 -1
  28. package/dist/address-stopwords.mjs +0 -84
  29. package/dist/address-stopwords.mjs.map +0 -1
  30. package/dist/allow-list.mjs +0 -196
  31. package/dist/allow-list.mjs.map +0 -1
  32. package/dist/clause-noun-heads.mjs +0 -75
  33. package/dist/clause-noun-heads.mjs.map +0 -1
  34. package/dist/common-words-en.mjs +0 -9887
  35. package/dist/common-words-en.mjs.map +0 -1
  36. package/dist/coreference.cs.mjs +0 -14
  37. package/dist/coreference.cs.mjs.map +0 -1
  38. package/dist/coreference.de.mjs +0 -14
  39. package/dist/coreference.de.mjs.map +0 -1
  40. package/dist/coreference.en.mjs +0 -14
  41. package/dist/coreference.en.mjs.map +0 -1
  42. package/dist/coreference.es.mjs +0 -22
  43. package/dist/coreference.es.mjs.map +0 -1
  44. package/dist/coreference.fr.mjs +0 -32
  45. package/dist/coreference.fr.mjs.map +0 -1
  46. package/dist/coreference.it.mjs +0 -27
  47. package/dist/coreference.it.mjs.map +0 -1
  48. package/dist/coreference.pl.mjs +0 -27
  49. package/dist/coreference.pl.mjs.map +0 -1
  50. package/dist/coreference.pt-br.mjs +0 -14
  51. package/dist/coreference.pt-br.mjs.map +0 -1
  52. package/dist/coreference.sk.mjs +0 -27
  53. package/dist/coreference.sk.mjs.map +0 -1
  54. package/dist/currencies.mjs +0 -231
  55. package/dist/currencies.mjs.map +0 -1
  56. package/dist/date-months.mjs +0 -618
  57. package/dist/date-months.mjs.map +0 -1
  58. package/dist/document-structure-headings.mjs +0 -90
  59. package/dist/document-structure-headings.mjs.map +0 -1
  60. package/dist/generic-roles.mjs +0 -244
  61. package/dist/generic-roles.mjs.map +0 -1
  62. package/dist/hotword-rules.mjs +0 -149
  63. package/dist/hotword-rules.mjs.map +0 -1
  64. package/dist/legal-form-leading-clauses.mjs +0 -23
  65. package/dist/legal-form-leading-clauses.mjs.map +0 -1
  66. package/dist/legal-forms.mjs +0 -2115
  67. package/dist/legal-forms.mjs.map +0 -1
  68. package/dist/legal-role-heads.cs.mjs +0 -42
  69. package/dist/legal-role-heads.cs.mjs.map +0 -1
  70. package/dist/legal-role-heads.de.mjs +0 -33
  71. package/dist/legal-role-heads.de.mjs.map +0 -1
  72. package/dist/legal-role-heads.en.mjs +0 -37
  73. package/dist/legal-role-heads.en.mjs.map +0 -1
  74. package/dist/legal-role-heads.es.mjs +0 -54
  75. package/dist/legal-role-heads.es.mjs.map +0 -1
  76. package/dist/legal-role-heads.fr.mjs +0 -72
  77. package/dist/legal-role-heads.fr.mjs.map +0 -1
  78. package/dist/legal-role-heads.it.mjs +0 -68
  79. package/dist/legal-role-heads.it.mjs.map +0 -1
  80. package/dist/legal-role-heads.pl.mjs +0 -84
  81. package/dist/legal-role-heads.pl.mjs.map +0 -1
  82. package/dist/legal-role-heads.pt-br.mjs +0 -63
  83. package/dist/legal-role-heads.pt-br.mjs.map +0 -1
  84. package/dist/legal-role-heads.sk.mjs +0 -80
  85. package/dist/legal-role-heads.sk.mjs.map +0 -1
  86. package/dist/manifest.mjs +0 -69
  87. package/dist/manifest.mjs.map +0 -1
  88. package/dist/names-exclusions.mjs +0 -223
  89. package/dist/names-exclusions.mjs.map +0 -1
  90. package/dist/names-first.mjs +0 -418
  91. package/dist/names-first.mjs.map +0 -1
  92. package/dist/names-nw-ar.mjs +0 -202
  93. package/dist/names-nw-ar.mjs.map +0 -1
  94. package/dist/names-nw-excluded-allcaps.mjs +0 -112
  95. package/dist/names-nw-excluded-allcaps.mjs.map +0 -1
  96. package/dist/names-nw-fil.mjs +0 -202
  97. package/dist/names-nw-fil.mjs.map +0 -1
  98. package/dist/names-nw-id.mjs +0 -210
  99. package/dist/names-nw-id.mjs.map +0 -1
  100. package/dist/names-nw-in.mjs +0 -526
  101. package/dist/names-nw-in.mjs.map +0 -1
  102. package/dist/names-nw-ja-latn.mjs +0 -260
  103. package/dist/names-nw-ja-latn.mjs.map +0 -1
  104. package/dist/names-nw-ko.mjs +0 -162
  105. package/dist/names-nw-ko.mjs.map +0 -1
  106. package/dist/names-nw-th.mjs +0 -188
  107. package/dist/names-nw-th.mjs.map +0 -1
  108. package/dist/names-nw-vi.mjs +0 -151
  109. package/dist/names-nw-vi.mjs.map +0 -1
  110. package/dist/names-nw-zh-latn.mjs +0 -197
  111. package/dist/names-nw-zh-latn.mjs.map +0 -1
  112. package/dist/names-surnames.mjs +0 -113
  113. package/dist/names-surnames.mjs.map +0 -1
  114. package/dist/names-title-tokens.mjs +0 -40
  115. package/dist/names-title-tokens.mjs.map +0 -1
  116. package/dist/person-stopwords.mjs +0 -205
  117. package/dist/person-stopwords.mjs.map +0 -1
  118. package/dist/section-headings.mjs +0 -64
  119. package/dist/section-headings.mjs.map +0 -1
  120. package/dist/sentence-verb-indicators.mjs +0 -232
  121. package/dist/sentence-verb-indicators.mjs.map +0 -1
  122. package/dist/signing-clauses.mjs +0 -78
  123. package/dist/signing-clauses.mjs.map +0 -1
  124. package/dist/stopwords.mjs +0 -9915
  125. package/dist/stopwords.mjs.map +0 -1
  126. package/dist/structural-single-cap-prefixes.mjs +0 -99
  127. package/dist/structural-single-cap-prefixes.mjs.map +0 -1
  128. package/dist/triggers.cs.mjs +0 -569
  129. package/dist/triggers.cs.mjs.map +0 -1
  130. package/dist/triggers.de.mjs +0 -139
  131. package/dist/triggers.de.mjs.map +0 -1
  132. package/dist/triggers.en.mjs +0 -119
  133. package/dist/triggers.en.mjs.map +0 -1
  134. package/dist/triggers.es.mjs +0 -96
  135. package/dist/triggers.es.mjs.map +0 -1
  136. package/dist/triggers.fr.mjs +0 -275
  137. package/dist/triggers.fr.mjs.map +0 -1
  138. package/dist/triggers.global.mjs +0 -79
  139. package/dist/triggers.global.mjs.map +0 -1
  140. package/dist/triggers.hu.mjs +0 -41
  141. package/dist/triggers.hu.mjs.map +0 -1
  142. package/dist/triggers.it.mjs +0 -74
  143. package/dist/triggers.it.mjs.map +0 -1
  144. package/dist/triggers.pl.mjs +0 -271
  145. package/dist/triggers.pl.mjs.map +0 -1
  146. package/dist/triggers.pt-br.mjs +0 -193
  147. package/dist/triggers.pt-br.mjs.map +0 -1
  148. package/dist/triggers.ro.mjs +0 -59
  149. package/dist/triggers.ro.mjs.map +0 -1
  150. package/dist/triggers.sk.mjs +0 -555
  151. package/dist/triggers.sk.mjs.map +0 -1
  152. package/dist/triggers.sv.mjs +0 -58
  153. package/dist/triggers.sv.mjs.map +0 -1
  154. package/dist/year-words.mjs +0 -62
  155. package/dist/year-words.mjs.map +0 -1
package/dist/index.d.mts CHANGED
@@ -1,728 +1,8 @@
1
1
  import { a as OPERATOR_TYPES, i as DetectionSource, n as DETECTION_SOURCES, o as OperatorType, r as DETECTOR_PRIORITY, t as DEFAULT_ENTITY_LABELS } from "./constants2.mjs";
2
- import { Match, PatternEntry, TextSearch } from "@stll/text-search";
3
- import { Validator } from "@stll/stdnum";
4
- import { Tokenizer } from "@huggingface/tokenizers";
2
+ import { $ as OperatorConfig, A as createNativePipelineFromPackage, C as SharedNativeRedactTextJsonOptions, D as assertNativeBindingVersion, E as SharedNativeSearchPackageOptions, F as getNativeBindingVersion, G as AnonymisationOperator, J as DenyListCategory, K as CustomDenyListEntry, N as encodeNativeSearchConfig, O as createNativeAnonymizerFromConfig, P as encodeNativeSearchConfigInput, Q as GazetteerEntry, S as SharedNativePreparedPackageOptions, T as SharedNativeRedactTextStreamJsonOptions, X as DictionaryMeta, Y as Dictionaries, Z as Entity, _ as PreparedNativeAnonymizer, a as NativeDiagnosticsBatchCallback, at as TriggerGroupConfig, b as SharedNativeDiagnosticsJsonOptions, c as NativePipelineEntity, ct as TriggerValidation, d as NativeRedactionResult, et as PipelineConfig, f as NativeResultEventCallback, g as PreparedAnonymizer, h as NativeStaticRedactionResult, i as NativeBindingVersionOptions, it as TriggerExtension, k as createNativeAnonymizerFromPackage, l as NativePipelineFromPackageOptions, lt as NativePreparedSearchConfig, m as NativeSearchPackageOptions, n as NativeAnonymizerFromConfigOptions, nt as ReviewDecision, o as NativeNormalizeOptions, ot as TriggerRule, p as NativeSearchPackageInput, q as CustomRegexPattern, r as NativeAnonymizerFromPackageOptions, rt as ReviewedEntity, s as NativeOperatorConfig, st as TriggerStrategy, t as NativeAnonymizeBinding, tt as RedactionResult, u as NativePreparedSearchBinding, v as PreparedNativePipeline, w as SharedNativeRedactTextOptions, x as SharedNativeDiagnosticsStreamJsonOptions, y as PreparedSearch, z as prepareNativeSearchPackage } from "./native.mjs";
3
+ import { A as readDefaultNativePipelinePackageFileAsync, B as redact_text_stream_json, C as native_package_version, D as preload_default_native_pipeline, E as preloadDefaultNativePipelineAsync, F as redactDefaultTextJson, G as NativePipelinePackageOptions, H as DEFAULT_NATIVE_PIPELINE_CONFIG, I as redact_default_text, J as createNativePipelineFromConfig, K as NativePipelineUnsupportedFeature, L as redact_default_text_json, M as readNativePipelinePackageFileAsync, N as read_default_native_pipeline_package_file, O as prepare_search_package, P as redactDefaultText, R as redact_text, S as load_prepared_package_file, T as preloadDefaultNativePipeline, U as NativePipelineBuildOptions, V as summary_diagnostics_json, W as NativePipelineCompatibility, X as prepareNativePipelineConfig, Y as getNativePipelineCompatibility, Z as prepareNativePipelinePackage, _ as diagnostics_stream_json, a as LoadNativeBindingOptions, b as loadNativeAnonymizeBinding, c as NativeRequire, d as availableDefaultNativePipelineLanguages, f as available_default_native_pipeline_languages, g as diagnostics_json, h as create_native_pipeline_from_default_package, i as DefaultNativePipelineWarmup, j as readNativePipelinePackageFile, k as readDefaultNativePipelinePackageFile, l as NativeSdkOptions, m as createNativePipelineFromPackageFile, n as DefaultNativePipelinePackageFileOptions, o as NativeLibc, p as createNativePipelineFromDefaultPackage, q as assertNativePipelineSupported, r as DefaultNativePipelinePackageOptions, s as NativePipelinePackageFileOptions, t as DEFAULT_NATIVE_PIPELINE_WARMUPS, u as NativeSdkPackageOptions, v as getDefaultNativePipeline, w as normalize_for_search, x as load_prepared_package, y as get_default_native_pipeline, z as redact_text_json } from "./native-node.mjs";
5
4
 
6
- //#region src/types.d.ts
7
- /**
8
- * Fields shared by every entity span in the source text.
9
- */
10
- type EntityBase = {
11
- start: number;
12
- end: number;
13
- label: string;
14
- text: string;
15
- score: number;
16
- sourceDetail?: "custom-deny-list" | "custom-regex" | "gazetteer-extension";
17
- };
18
- /**
19
- * A PII entity span found by a primary detection layer
20
- * (regex, NER, legal forms, deny list, ...).
21
- */
22
- type DetectedEntity = EntityBase & {
23
- source: Exclude<DetectionSource, typeof DETECTION_SOURCES.COREFERENCE>;
24
- };
25
- /**
26
- * An alias mention of a previously detected entity: a
27
- * defined term ("the Seller") or a propagated bare
28
- * mention ("Acme" after "Acme Corp.").
29
- *
30
- * `corefSourceText` is required by construction, so an
31
- * alias cannot exist without the link back to its source
32
- * entity. Placeholder numbering reads it to give the
33
- * alias the same placeholder as the source. The link
34
- * travels with the entity instead of living in a
35
- * side-channel map that a producer could forget to
36
- * write — or that a later pass could clear.
37
- */
38
- type CorefAliasEntity = EntityBase & {
39
- source: typeof DETECTION_SOURCES.COREFERENCE; /** Full text of the source entity this alias refers to. */
40
- corefSourceText: string;
41
- };
42
- /**
43
- * A detected PII entity span in the source text.
44
- * Every detection layer produces these.
45
- */
46
- type Entity = DetectedEntity | CorefAliasEntity;
47
- /**
48
- * Entity after human review. Extends the base Entity
49
- * with a review decision.
50
- */
51
- type ReviewDecision = "confirmed" | "rejected" | "relabeled";
52
- type ReviewedEntity = Entity & {
53
- decision?: ReviewDecision;
54
- originalLabel?: string;
55
- };
56
- /**
57
- * A single entry in the workspace-scoped gazetteer
58
- * (deny list). Persisted in IndexedDB.
59
- */
60
- type GazetteerEntry = {
61
- id: string;
62
- canonical: string;
63
- label: string;
64
- variants: string[];
65
- workspaceId: string;
66
- createdAt: number;
67
- source: "manual" | "confirmed-from-model";
68
- };
69
- /** Extraction strategy — closed discriminated union. */
70
- type TriggerStrategy = {
71
- type: "to-next-comma";
72
- /**
73
- * Optional list of lowercase keywords that terminate
74
- * the value scan, in addition to commas/newlines. Useful
75
- * for triggers like court names that may continue past
76
- * a missing comma into adjacent clause text ("Městským
77
- * soudem v Praze dne 1. 1. 2020"); listing `"dne"` here
78
- * stops the scan at the date boundary. Matched on a
79
- * word-boundary, case-insensitive.
80
- */
81
- stopWords?: string[];
82
- /**
83
- * Hard cap on the captured span length, in characters,
84
- * regardless of where the next comma / stop char sits.
85
- * Use for triggers that label short formulaic phrases
86
- * ("State of Delaware") and must not absorb the rest
87
- * of a long forum-selection clause when the comma is
88
- * sentences away. Falls back to the default 100-char
89
- * fallback when omitted.
90
- */
91
- maxLength?: number;
92
- } | {
93
- type: "to-end-of-line";
94
- } | {
95
- type: "n-words";
96
- count: number;
97
- } | {
98
- type: "company-id-value";
99
- } | {
100
- type: "address";
101
- maxChars?: number;
102
- } | {
103
- /**
104
- * Extract the first regex match in the value text.
105
- * Useful for shape-bounded values that follow a
106
- * label on the same line as other fields, where
107
- * `to-end-of-line` would over-capture. The pattern
108
- * is anchored to the start of the (already
109
- * leading-whitespace-stripped) value, so use
110
- * `(?:.*?)` prefix only when intentional.
111
- */
112
- type: "match-pattern";
113
- pattern: string;
114
- flags?: string;
115
- };
116
- /** Validation rules — closed discriminated union. */
117
- type TriggerValidation = {
118
- type: "starts-uppercase";
119
- } | {
120
- type: "min-length";
121
- min: number;
122
- } | {
123
- type: "max-length";
124
- max: number;
125
- } | {
126
- type: "no-digits";
127
- } | {
128
- type: "has-digits";
129
- } | {
130
- type: "matches-pattern";
131
- pattern: string;
132
- flags?: string;
133
- }
134
- /**
135
- * Run a named stdnum validator (checksum + length)
136
- * against the captured value. Keeps the trigger
137
- * path symmetrical with the formatted-regex
138
- * detectors so e.g. `CPF nº 00000000000` does not
139
- * survive as a tax-ID entity.
140
- */
141
- | {
142
- type: "valid-id";
143
- validator: ValidIdValidator;
144
- };
145
- /** Built-in stdnum validators that can be referenced
146
- * by `valid-id` validations. */
147
- type ValidIdValidator = "br.cpf" | "br.cnpj" | "us.rtn";
148
- /** Auto-generated trigger variants — closed set. */
149
- type TriggerExtension = "add-colon" | "add-trailing-space" | "add-colon-space" | "normalize-spaces";
150
- /** V2 trigger config entry (JSON shape). */
151
- type TriggerGroupConfig = {
152
- id?: string;
153
- triggers: string[];
154
- label: string;
155
- strategy: TriggerStrategy;
156
- extensions?: TriggerExtension[];
157
- validations?: TriggerValidation[];
158
- /** When true, include the trigger text in the
159
- * entity span (e.g., court names). */
160
- includeTrigger?: boolean;
161
- };
162
- /** Compiled validation with pre-built regex. */
163
- type CompiledValidation = {
164
- type: "starts-uppercase";
165
- re: RegExp;
166
- } | {
167
- type: "min-length";
168
- min: number;
169
- } | {
170
- type: "max-length";
171
- max: number;
172
- } | {
173
- type: "no-digits";
174
- re: RegExp;
175
- } | {
176
- type: "has-digits";
177
- re: RegExp;
178
- } | {
179
- type: "matches-pattern";
180
- re: RegExp;
181
- } | {
182
- type: "valid-id";
183
- check: (value: string) => boolean;
184
- };
185
- /**
186
- * Runtime rule — one per trigger string after
187
- * expansion. Fed to the Aho-Corasick automaton.
188
- */
189
- type TriggerRule = {
190
- trigger: string;
191
- label: string;
192
- strategy: TriggerStrategy;
193
- validations: CompiledValidation[];
194
- includeTrigger: boolean;
195
- };
196
- /** Per-label operator selection. Key is the entity label. */
197
- type OperatorConfig = {
198
- /** Operator per label. Missing labels default to "replace". */operators: Record<string, OperatorType>; /** Custom replacement string for the redact operator. */
199
- redactString: string;
200
- };
201
- /** Whether an operator produces a reversible redaction entry. */
202
- type OperatorReversibility = "reversible" | "irreversible";
203
- type AnonymisationOperator = {
204
- type: OperatorType;
205
- reversibility: OperatorReversibility;
206
- /**
207
- * Apply the operator to a single entity occurrence.
208
- * Returns the replacement string to embed in the document.
209
- */
210
- apply: (text: string, label: string, placeholder: string, redactString: string) => string;
211
- };
212
- /**
213
- * Redacted document output with stable entity mapping.
214
- */
215
- type RedactionResult = {
216
- redactedText: string;
217
- /**
218
- * Maps placeholder to original text. Only populated for
219
- * reversible operators (replace). Empty for redact.
220
- */
221
- redactionMap: Map<string, string>; /** Maps placeholder to the operator that produced it. */
222
- operatorMap: Map<string, OperatorType>;
223
- entityCount: number;
224
- };
225
- /**
226
- * Configuration for the detection pipeline.
227
- */
228
- type DenyListCategory = "Names" | "Places" | "Addresses" | "Courts" | "Financial" | "Government" | "Healthcare" | "Education" | "Political" | "Organizations" | "International";
229
- /**
230
- * Metadata for a single dictionary entry in the
231
- * deny-list system. Mirrors the shape from
232
- * the anonymize-data package so consumers can pass
233
- * pre-loaded data without a runtime dependency.
234
- */
235
- type DictionaryMeta = {
236
- label: string;
237
- category: DenyListCategory;
238
- country: string | null;
239
- };
240
- /**
241
- * Caller-supplied exact terms for deny-list matching.
242
- * These entries are merged with the published deny-list
243
- * dictionaries when `enableDenyList` is enabled.
244
- */
245
- type CustomDenyListEntry = {
246
- value: string;
247
- label: string;
248
- variants?: readonly string[];
249
- };
250
- /**
251
- * Caller-supplied regex detector. The pattern is passed
252
- * to the underlying text-search regex engine, so use its
253
- * supported regex syntax. Inline flags such as `(?i)` are
254
- * accepted when supported by that engine.
255
- */
256
- type CustomRegexPattern = {
257
- pattern: string;
258
- label: string;
259
- score?: number;
260
- };
261
- /**
262
- * Pre-loaded dictionary data for dependency injection.
263
- * Consumers that want name/city/deny-list detection
264
- * load dictionaries themselves (e.g. from the
265
- * anonymize-data package) and pass them here; the
266
- * anonymize package has zero cross-package imports.
267
- *
268
- * All fields are optional. When a field is absent,
269
- * the corresponding detection path is skipped (same
270
- * behavior as when no dictionaries are available).
271
- */
272
- type Dictionaries = {
273
- /**
274
- * First names per language code (e.g., "cs", "de").
275
- * Merged with legacy config names at init time.
276
- */
277
- firstNames?: Readonly<Record<string, readonly string[]>>;
278
- /**
279
- * Surnames per language code.
280
- * Merged with legacy config names at init time.
281
- */
282
- surnames?: Readonly<Record<string, readonly string[]>>;
283
- /**
284
- * Non-Western name tokens per locale code
285
- * (e.g., "in", "ar", "ja-latn", "ko", "zh-latn",
286
- * "th", "vi", "fil", "id"). Merged with bundled
287
- * names-nw-*.json data at init time.
288
- */
289
- nonWesternNames?: Readonly<Record<string, readonly string[]>>;
290
- /**
291
- * Pre-loaded deny-list dictionaries keyed by
292
- * dictionary ID (e.g., "courts/CZ", "banks/DE").
293
- * Each value is the array of terms for that
294
- * dictionary.
295
- */
296
- denyList?: Readonly<Record<string, readonly string[]>>;
297
- /**
298
- * Metadata per dictionary ID. Required when
299
- * `denyList` is provided so the pipeline knows
300
- * labels, categories, and country filters.
301
- */
302
- denyListMeta?: Readonly<Record<string, DictionaryMeta>>;
303
- /**
304
- * Pre-loaded city names, already merged across
305
- * all desired countries.
306
- *
307
- * Prefer `citiesByCountry` when callers also pass
308
- * `denyListCountries` / `denyListRegions`; merged
309
- * city arrays cannot be scoped after injection.
310
- */
311
- cities?: readonly string[];
312
- /**
313
- * Pre-loaded city names keyed by ISO 3166-1 alpha-2
314
- * country code. When provided, the deny-list builder
315
- * applies `denyListCountries` / `denyListRegions`
316
- * before adding city patterns to the search automaton.
317
- */
318
- citiesByCountry?: Readonly<Record<string, readonly string[]>>;
319
- };
320
- type PipelineConfig = {
321
- threshold: number;
322
- enableTriggerPhrases: boolean;
323
- enableRegex: boolean;
324
- /**
325
- * Enables legal-form organization detection.
326
- * Required for typed callers; legacy untyped
327
- * callers that omit this field are treated as
328
- * enabled at runtime for backward compatibility.
329
- */
330
- enableLegalForms: boolean;
331
- /**
332
- * Enables first-name/surname/title corpus matching.
333
- * When deny-list mode is enabled, this also controls
334
- * whether name-corpus entries are injected into the
335
- * deny-list search automaton.
336
- */
337
- enableNameCorpus: boolean;
338
- /**
339
- * Optional language scope for first-name/surname
340
- * dictionaries, using the keys present in
341
- * `dictionaries.firstNames` / `dictionaries.surnames`
342
- * (for example `["en", "de"]`). When omitted, all
343
- * injected name languages are used for backward
344
- * compatibility.
345
- */
346
- nameCorpusLanguages?: string[];
347
- enableDenyList: boolean;
348
- denyListCountries?: string[];
349
- denyListRegions?: string[];
350
- denyListExcludeCategories?: string[];
351
- /**
352
- * Caller-owned exact terms to match through the
353
- * deny-list layer. Requires `enableDenyList: true`.
354
- */
355
- customDenyList?: readonly CustomDenyListEntry[];
356
- /**
357
- * Caller-owned regex detectors. Requires
358
- * `enableRegex: true`.
359
- */
360
- customRegexes?: readonly CustomRegexPattern[];
361
- enableGazetteer: boolean;
362
- /**
363
- * Detect country names (ISO 3166-1 names, curated
364
- * aliases, alpha-3 codes). Defaults to true. Names
365
- * span all manifest languages plus widely-used
366
- * additions (Dutch, Russian, Chinese, Arabic, etc.).
367
- */
368
- enableCountries?: boolean;
369
- enableNer: boolean;
370
- enableConfidenceBoost: boolean;
371
- enableCoreference: boolean;
372
- enableZoneClassification?: boolean;
373
- enableHotwordRules?: boolean;
374
- /**
375
- * Requested output labels. An empty array means
376
- * "do not filter by label" for deterministic
377
- * detectors; NER falls back to DEFAULT_ENTITY_LABELS.
378
- */
379
- labels: string[];
380
- workspaceId: string;
381
- /**
382
- * Pre-loaded dictionary data for name, deny-list,
383
- * and city detection. When omitted, dictionary-based
384
- * detection paths are skipped. Consumers load from
385
- * the anonymize-data package and pass the data here.
386
- */
387
- dictionaries?: Dictionaries;
388
- };
389
- //#endregion
390
- //#region src/detectors/regex.d.ts
391
- type RegexMeta = {
392
- label: string;
393
- score: number;
394
- sourceDetail?: Entity["sourceDetail"]; /** Post-match stdnum validator for confirmation. */
395
- validator?: Validator; /** Extract the identifier portion when context is part of the regex span. */
396
- validatorInput?: (text: string) => string;
397
- };
398
- /** Flat pattern array for text-search. */
399
- declare const REGEX_PATTERNS: readonly string[];
400
- /** Parallel metadata. Index = pattern index. */
401
- declare const REGEX_META: readonly RegexMeta[];
402
- /**
403
- * Get dynamically built date patterns from
404
- * date-months.json. Returns a cached promise; the JSON
405
- * is loaded only once.
406
- */
407
- declare const getDatePatterns: () => Promise<string[]>;
408
- /** Date pattern metadata (all are score 1 dates). */
409
- declare const DATE_PATTERN_META: Readonly<RegexMeta>;
410
- /**
411
- * Get dynamically built monetary amount patterns from
412
- * currencies.json. Returns a cached promise; the JSON
413
- * is loaded only once.
414
- */
415
- declare const getCurrencyPatterns: () => Promise<string[]>;
416
- /** Currency pattern metadata (score 0.9). */
417
- declare const CURRENCY_PATTERN_META: Readonly<RegexMeta>;
418
- /**
419
- * Process regex matches from the unified search.
420
- * Receives all matches; filters to the regex slice
421
- * via sliceStart/sliceEnd. Local index into META is
422
- * match.pattern - sliceStart.
423
- *
424
- * For stdnum-derived patterns (those with a validator
425
- * in META), the matched text is passed through the
426
- * validator's validate() method. If validation fails,
427
- * the match is discarded as a false positive.
428
- */
429
- declare const processRegexMatches: (allMatches: Match[], sliceStart: number, sliceEnd: number, meta_: readonly RegexMeta[]) => Entity[];
430
- //#endregion
431
- //#region src/detectors/deny-list.d.ts
432
- type DenyListConfig = Pick<PipelineConfig, "enableDenyList" | "enableNameCorpus" | "nameCorpusLanguages" | "denyListCountries" | "denyListRegions" | "denyListExcludeCategories" | "customDenyList" | "dictionaries" | "enableCountries">;
433
- /**
434
- * Source tag for each pattern in the automaton.
435
- * "deny-list" = standard deny list entry
436
- * "city" = city dictionary entry
437
- * "custom-deny-list" = caller-owned exact term
438
- * "first-name" = name corpus first name
439
- * "surname" = name corpus surname
440
- * "title" = academic/professional title
441
- */
442
- type PatternSource = "deny-list" | "city" | "custom-deny-list" | "first-name" | "surname" | "title";
443
- type PatternLabels = string | string[];
444
- type PatternSources = PatternSource | PatternSource[];
445
- /**
446
- * Pre-built deny list data. Constructed once by
447
- * `buildDenyList`, reused across `processDenyListMatches`
448
- * calls. Contains PatternEntry[] for the unified builder
449
- * plus parallel label/source arrays for post-processing.
450
- */
451
- type DenyListData = {
452
- /**
453
- * Maps pattern index → entity labels (plural).
454
- * Same pattern can have multiple labels when it
455
- * appears in multiple dictionaries (e.g., "Denver"
456
- * is both a person name and a city name).
457
- */
458
- labels: PatternLabels[]; /** Maps pattern index → labels contributed by custom entries. */
459
- customLabels: (PatternLabels | undefined)[]; /** Maps pattern index → original pattern text. */
460
- originals: string[]; /** Maps pattern index → source types (plural). */
461
- sources: PatternSources[];
462
- };
463
- /**
464
- * Resolve which dictionaries to load based on country
465
- * and category filters, then build the deny list data.
466
- * The returned data provides PatternEntry[] for the
467
- * unified builder and parallel arrays for
468
- * post-processing.
469
- *
470
- * Dictionary data is injected via `config.dictionaries`.
471
- * Returns null if no dictionaries are provided.
472
- */
473
- declare const buildDenyList: (config: DenyListConfig, ctx?: PipelineContext) => Promise<DenyListData | null>;
474
- /**
475
- * Ensure all deny-list support data (stopwords, allow
476
- * list, person stopwords, generic roles) is loaded on
477
- * the given context. Call this before
478
- * processDenyListMatches / filterFalsePositives when
479
- * the search instance was built on a different context
480
- * (e.g. cachedSearch).
481
- */
482
- declare const ensureDenyListData: (ctx?: PipelineContext, dictionaries?: Dictionaries, nameCorpusLanguages?: readonly string[]) => Promise<void>;
483
- /**
484
- * Process deny list matches from the unified search.
485
- * Receives all matches; filters to the deny list slice
486
- * via sliceStart/sliceEnd. Local index into data.labels,
487
- * data.originals, data.sources is match.pattern - sliceStart.
488
- *
489
- * Two-pass approach to reduce false positives:
490
- * 1. Collect all matches (case-insensitive,
491
- * whole-word via Rust automaton)
492
- * 2. Require uppercase start in source text
493
- * 3. For person names, require at least one
494
- * mid-sentence occurrence to prove proper noun
495
- * 4. Return all occurrences of validated terms
496
- */
497
- declare const processDenyListMatches: (allMatches: Match[], sliceStart: number, sliceEnd: number, fullText: string, data: DenyListData, ctx?: PipelineContext) => Entity[];
498
- //#endregion
499
- //#region src/detectors/countries.d.ts
500
- /**
501
- * Pre-built country patterns + parallel label/source
502
- * metadata. Constructed once and reused across pipeline
503
- * runs.
504
- */
505
- type CountryData = {
506
- /** Maps local pattern index to entity label. Always "country". */labels: string[];
507
- /**
508
- * Maps local pattern index to the alpha-2 ISO code the
509
- * pattern resolves to. Used for downstream coreference /
510
- * placeholder grouping.
511
- */
512
- isoCodes: string[]; /** Maps local pattern index to pattern variant kind. */
513
- variants: CountryVariant[];
514
- };
515
- type CountryVariant = "name" | "alias" | "alpha3" | "alpha2";
516
- //#endregion
517
- //#region src/build-unified-search.d.ts
518
- type PatternSlice = {
519
- start: number;
520
- end: number;
521
- };
522
- type GazetteerData = {
523
- /** Maps local pattern index to entry label. */labels: string[];
524
- /**
525
- * Whether each pattern is fuzzy (distance > 0).
526
- * Used by the post-processor to assign scores.
527
- */
528
- isFuzzy: boolean[];
529
- };
530
- type UnifiedSearchInstance = {
531
- /** Regex + triggers + legal-forms. */tsRegex: TextSearch; /** Caller-owned custom regexes, isolated for overlap preservation. */
532
- tsCustomRegex: TextSearch; /** Deny-list + street-types + gazetteer. */
533
- tsLiterals: TextSearch;
534
- slices: {
535
- regex: PatternSlice;
536
- customRegex: PatternSlice;
537
- legalForms: PatternSlice;
538
- triggers: PatternSlice;
539
- denyList: PatternSlice;
540
- streetTypes: PatternSlice;
541
- gazetteer: PatternSlice;
542
- countries: PatternSlice;
543
- };
544
- regexMeta: readonly RegexMeta[];
545
- customRegexMeta: readonly RegexMeta[];
546
- triggerRules: readonly TriggerRule[];
547
- denyListData: DenyListData | null;
548
- gazetteerData: GazetteerData | null;
549
- countryData: CountryData | null;
550
- };
551
- declare const buildUnifiedSearch: (config: PipelineConfig, gazetteerEntries?: GazetteerEntry[], ctx?: PipelineContext) => Promise<UnifiedSearchInstance>;
552
- //#endregion
553
- //#region src/context.d.ts
554
- /**
555
- * Build a stable cache key for an entity that survives
556
- * shallow copies (spread). Uses position + label so the
557
- * key is identical for the original object and any
558
- * `{ ...entity }` copy produced by mergeAndDedup.
559
- *
560
- * @deprecated No longer used internally: coref alias
561
- * links travel on the entities themselves
562
- * (`corefSourceText`). Kept for API compatibility.
563
- */
564
- declare const corefKey: (e: Entity) => string;
565
- /**
566
- * Compiled RegExp pattern used for coreference
567
- * definition extraction.
568
- */
569
- type DefinitionPattern = {
570
- pattern: RegExp;
571
- };
572
- /**
573
- * Cached data for the name corpus detector.
574
- * Populated by initNameCorpus; consumed by
575
- * detectNameCorpus and deny-list AC integration.
576
- */
577
- type NameCorpusData = {
578
- firstNames: ReadonlySet<string>;
579
- surnames: ReadonlySet<string>;
580
- titleTokens: ReadonlySet<string>;
581
- /** Abbreviation-style titles whose trailing dot is
582
- * part of the title, not a sentence boundary.
583
- * Contains the lowercase, dot-stripped form
584
- * (e.g., "dr", "smt", "atty"). */
585
- titleAbbreviations: ReadonlySet<string>;
586
- excludedWords: ReadonlySet<string>;
587
- /** Lowercased common English words. A name chain whose
588
- * every token is a common word (e.g. "Loan Documents",
589
- * where "Loan" coincides with a Vietnamese given name)
590
- * is treated as a common-word phrase, not a person. */
591
- commonWords: ReadonlySet<string>; /** Non-Western name tokens merged across all locales. */
592
- nonWesternNames: ReadonlySet<string>; /** All-caps acronyms excluded from name detection. */
593
- excludedAllCaps: ReadonlySet<string>; /** Raw arrays exposed for deny-list AC integration. */
594
- firstNamesList: readonly string[];
595
- surnamesList: readonly string[];
596
- titlesList: readonly string[];
597
- excludedList: readonly string[];
598
- nonWesternNamesList: readonly string[];
599
- excludedAllCapsList: readonly string[];
600
- };
601
- /**
602
- * All cached state for a single pipeline run (or
603
- * sequence of runs sharing the same config). Replacing
604
- * module-level singletons with this object enables
605
- * concurrent pipelines with different configs and
606
- * simplifies testing.
607
- *
608
- * Each field starts null and is populated lazily on
609
- * first use by the corresponding loader function.
610
- */
611
- type PipelineContext = {
612
- search: UnifiedSearchInstance | null;
613
- searchKey: string;
614
- searchPromise: Promise<UnifiedSearchInstance> | null;
615
- nameCorpus: NameCorpusData | null;
616
- nameCorpusKey: string;
617
- nameCorpusPromise: Promise<void> | null;
618
- stopwords: ReadonlySet<string> | null;
619
- stopwordsPromise: Promise<ReadonlySet<string>> | null;
620
- allowList: ReadonlySet<string> | null;
621
- allowListPromise: Promise<ReadonlySet<string>> | null;
622
- personStopwords: ReadonlySet<string> | null;
623
- personStopwordsPromise: Promise<ReadonlySet<string>> | null;
624
- addressStopwords: ReadonlySet<string> | null;
625
- addressStopwordsPromise: Promise<ReadonlySet<string>> | null; /** First-name exclusions for stopword filtering. */
626
- firstNameExclusions: ReadonlySet<string> | null;
627
- firstNameExclusionCorpusLen: number;
628
- genericRoles: ReadonlySet<string> | null;
629
- genericRolesPromise: Promise<ReadonlySet<string>> | null;
630
- corefPatterns: DefinitionPattern[] | null;
631
- corefPatternsPromise: Promise<DefinitionPattern[]> | null;
632
- corefLoadAttempted: boolean;
633
- roleStopSet: ReadonlySet<string> | null;
634
- roleStopSetPromise: Promise<ReadonlySet<string>> | null;
635
- zoneHeadingPatterns: RegExp[] | null;
636
- zoneSigningPatterns: RegExp[] | null;
637
- zoneInitPromise: Promise<void> | null;
638
- };
639
- /** Create a fresh, empty pipeline context. */
640
- declare const createPipelineContext: () => PipelineContext;
641
- //#endregion
642
- //#region src/pipeline.d.ts
643
- /** Strip leading/trailing whitespace and punctuation. */
644
- declare const sanitizeEntities: (entities: Entity[]) => Entity[];
645
- declare const mergeAndDedup: (...layers: Entity[][]) => Entity[];
646
- type NerInferenceFn = (fullText: string, labels: string[], threshold: number, signal?: AbortSignal) => Promise<Entity[]>;
647
- type PipelineSearchOptions = {
648
- config: PipelineConfig;
649
- gazetteerEntries?: GazetteerEntry[];
650
- context?: PipelineContext;
651
- };
652
- /**
653
- * Pre-build and cache the unified search instance for a
654
- * pipeline configuration. Use the same context in
655
- * `runPipeline` to reuse the prepared automata without
656
- * passing `cachedSearch` around manually.
657
- */
658
- declare const preparePipelineSearch: ({
659
- config,
660
- gazetteerEntries,
661
- context
662
- }: PipelineSearchOptions) => Promise<UnifiedSearchInstance>;
663
- /**
664
- * Options for {@link runPipeline}.
665
- *
666
- * @property cachedSearch Pre-built search instance.
667
- * When provided, `config` and `gazetteerEntries`
668
- * are not used for building; the caller must
669
- * ensure the instance matches both parameters.
670
- */
671
- type PipelineOptions = {
672
- fullText: string;
673
- config: PipelineConfig;
674
- gazetteerEntries: GazetteerEntry[];
675
- nerInference?: NerInferenceFn | null;
676
- onProgress?: (step: string, detail: string) => void;
677
- cachedSearch?: UnifiedSearchInstance;
678
- signal?: AbortSignal;
679
- context?: PipelineContext;
680
- };
681
- /**
682
- * Run the full detection pipeline.
683
- *
684
- * Two TextSearch instances scan the text (regex +
685
- * literals). Results are dispatched to each
686
- * detector's post-processor by pattern index range.
687
- *
688
- * Pass an AbortSignal to cancel the pipeline between
689
- * stages. Throws a DOMException with name "AbortError"
690
- * when cancelled.
691
- *
692
- * Pass an optional `context` to isolate cached state
693
- * from other pipeline runs. If omitted, a module-level
694
- * default context is used (backward compatible).
695
- */
696
- declare const runPipeline: (options: PipelineOptions) => Promise<Entity[]>;
697
- //#endregion
698
5
  //#region src/redact.d.ts
699
- /**
700
- * Build a stable mapping from entity text to numbered
701
- * placeholders. Same real-world value always maps to the
702
- * same placeholder (e.g., "Dr. Muller" and "Dr. Muller"
703
- * both become [PERSON_1]).
704
- *
705
- * Placeholder format: [LABEL_N] where LABEL is uppercase
706
- * and N is a 1-based counter per label.
707
- *
708
- * @param _ctx Unused. Kept for signature compatibility;
709
- * coref alias links now travel on the entities
710
- * themselves (`corefSourceText`).
711
- */
712
- declare const buildPlaceholderMap: (entities: Entity[], _ctx?: PipelineContext) => Map<string, string>;
713
- /**
714
- * Apply redactions to the source text, replacing each
715
- * confirmed entity span using the configured operator.
716
- *
717
- * Co-references are consistent: if the same text appears
718
- * multiple times, all occurrences get the same placeholder.
719
- *
720
- * @param ctx Pipeline context. Must be the same instance
721
- * passed to `runPipeline` (or `findCoreferenceSpans`)
722
- * so coreference placeholder links are preserved.
723
- * Defaults to `defaultContext` for single-tenant usage.
724
- */
725
- declare const redactText: (fullText: string, entities: Entity[], config?: OperatorConfig, ctx?: PipelineContext) => RedactionResult;
726
6
  /**
727
7
  * Serialize the redaction key to JSON for export.
728
8
  * Includes operator metadata so the export is self-describing.
@@ -735,483 +15,5 @@ declare const exportRedactionKey: (redactionMap: Map<string, string>, operatorMa
735
15
  */
736
16
  declare const deanonymise: (redactedText: string, redactionMap: Map<string, string>) => string;
737
17
  //#endregion
738
- //#region src/operators.d.ts
739
- declare const OPERATOR_REGISTRY: {
740
- readonly replace: AnonymisationOperator;
741
- readonly redact: AnonymisationOperator;
742
- };
743
- /**
744
- * Default operator config: replace for all labels.
745
- * Preserves existing pipeline behaviour.
746
- */
747
- declare const DEFAULT_OPERATOR_CONFIG: OperatorConfig;
748
- /**
749
- * Resolve the operator for a label, falling back to "replace".
750
- */
751
- declare const resolveOperator: (config: OperatorConfig, label: string) => OperatorType;
752
- //#endregion
753
- //#region src/detectors/legal-forms.d.ts
754
- declare const warmLegalRoleHeads: () => Promise<void>;
755
- /**
756
- * Build legal form regex pattern strings.
757
- * Returns an array of regex strings for the unified
758
- * TextSearch builder. Empty if data package is not
759
- * installed.
760
- */
761
- declare const buildLegalFormPatterns: () => Promise<string[]>;
762
- /**
763
- * Process legal form matches from the unified search.
764
- * Receives all matches; filters to the legal forms
765
- * slice via sliceStart/sliceEnd.
766
- *
767
- * The role-head trimming step reads per-language data from
768
- * a cache that `runPipeline` warms via `warmLegalRoleHeads()`
769
- * before calling this. Callers that invoke
770
- * `processLegalFormMatches` directly (without going through
771
- * `runPipeline`) must `await warmLegalRoleHeads()` first;
772
- * otherwise the trim falls back to a no-op and sentence-
773
- * fragment fixes do not apply.
774
- */
775
- declare const processLegalFormMatches: (allMatches: Match[], sliceStart: number, sliceEnd: number, fullText?: string, options?: {
776
- suppressExtendBackward?: boolean;
777
- }) => Entity[];
778
- //#endregion
779
- //#region src/detectors/triggers.d.ts
780
- type TriggerPatterns = {
781
- patterns: string[];
782
- rules: TriggerRule[];
783
- };
784
- declare const buildTriggerPatterns: () => Promise<TriggerPatterns>;
785
- /**
786
- * Process trigger matches from the unified search.
787
- * Receives all matches; filters to the trigger slice
788
- * via sliceStart/sliceEnd. Uses fullText for value
789
- * extraction (the unified search runs on lowercased
790
- * text, but extraction needs original casing).
791
- */
792
- declare const processTriggerMatches: (allMatches: Match[], sliceStart: number, sliceEnd: number, fullText: string, rules: readonly TriggerRule[]) => Entity[];
793
- //#endregion
794
- //#region src/detectors/address-seeds.d.ts
795
- declare const buildStreetTypePatterns: () => Promise<string[]>;
796
- /**
797
- * Process address seeds from the unified search.
798
- * Receives all matches; filters to the street types
799
- * slice via sliceStart/sliceEnd. Uses fullText and
800
- * existingEntities for seed collection, clustering,
801
- * expansion, and scoring.
802
- *
803
- * Runs as a post-processor after all other detectors,
804
- * using their output as seed sources.
805
- */
806
- declare const processAddressSeeds: (allMatches: Match[], sliceStart: number, sliceEnd: number, fullText: string, existingEntities: Entity[]) => Promise<Entity[]>;
807
- //#endregion
808
- //#region src/detectors/gazetteer.d.ts
809
- /**
810
- * Build TextSearch-compatible patterns from gazetteer
811
- * entries. Returns:
812
- * - Exact literal patterns for all terms
813
- * - Fuzzy patterns (distance: 2) for terms >= 4 chars
814
- * - Parallel metadata arrays for post-processing
815
- *
816
- * Patterns are ordered: all exact first, then all
817
- * fuzzy. The isFuzzy array marks which are which.
818
- */
819
- declare const buildGazetteerPatterns: (entries: GazetteerEntry[]) => {
820
- patterns: PatternEntry[];
821
- data: GazetteerData;
822
- };
823
- /**
824
- * Process gazetteer matches from the unified literal
825
- * search. Receives all matches; filters to the
826
- * gazetteer slice via sliceStart/sliceEnd.
827
- *
828
- * Exact matches get score 0.9; fuzzy matches get
829
- * 0.85. Fuzzy matches that overlap an exact match
830
- * are dropped.
831
- *
832
- * For exact matches, attempts prefix extension for
833
- * legal suffixes ("a.s.", "GmbH", "s.r.o." after
834
- * the matched term).
835
- */
836
- declare const processGazetteerMatches: (allMatches: Match[], sliceStart: number, sliceEnd: number, fullText: string, data: GazetteerData) => Entity[];
837
- //#endregion
838
- //#region src/detectors/coreference.d.ts
839
- type DefinedTerm = {
840
- alias: string;
841
- label: string; /** Position of the definition in the source text */
842
- definitionStart: number; /** Original entity text the alias refers to */
843
- sourceText: string;
844
- };
845
- declare const extractDefinedTerms: (fullText: string, entities: Entity[], ctx?: PipelineContext) => Promise<DefinedTerm[]>;
846
- /**
847
- * Find all occurrences of defined-term aliases in the
848
- * full text. Returns Entity spans for each match.
849
- *
850
- * Respects word boundaries: "Kupující" must not match
851
- * inside "Kupujícímu". A match is valid only if the
852
- * character before the start and after the end are NOT
853
- * word characters (letter/digit).
854
- *
855
- * Each returned alias carries `corefSourceText` linking
856
- * it to its source entity text, for consistent
857
- * placeholder numbering.
858
- *
859
- * @param _ctx Unused. Kept for signature compatibility;
860
- * alias links now travel on the entities themselves.
861
- */
862
- declare const findCoreferenceSpans: (fullText: string, terms: DefinedTerm[], _ctx?: PipelineContext) => Entity[];
863
- //#endregion
864
- //#region src/detectors/org-propagation.d.ts
865
- /**
866
- * After the main detection pass, collect organization
867
- * entities with a legal form suffix, strip the suffix
868
- * to get the base name, and re-scan the full text for
869
- * bare mentions of that base name. Returns new entities
870
- * for occurrences not already covered.
871
- *
872
- * Propagated mentions are coref aliases: each carries
873
- * `corefSourceText` linking it to the full seed entity
874
- * text, so placeholder numbering assigns the bare
875
- * mention the same placeholder as its source ("Acme"
876
- * and "Acme Corp." both become [ORGANIZATION_1]).
877
- */
878
- declare const propagateOrgNames: (entities: Entity[], fullText: string) => Entity[];
879
- //#endregion
880
- //#region src/detectors/names.d.ts
881
- declare const getNameCorpusNonWesternNames: (ctx?: PipelineContext) => readonly string[];
882
- /**
883
- * Load name corpus data from injected dictionaries
884
- * and legacy config files. Merges all sources.
885
- *
886
- * Safe to call multiple times; only loads once per
887
- * context. Must be called before detectNameCorpus or
888
- * the getNameCorpus*() accessors are used.
889
- *
890
- * @param dictionaries Optional pre-loaded dictionaries
891
- * with per-language first names and surnames. When
892
- * omitted, only legacy config files are used.
893
- */
894
- declare const initNameCorpus: (ctx?: PipelineContext, dictionaries?: Dictionaries, languages?: readonly string[]) => Promise<void>;
895
- type NameCorpusDetectionOptions = {
896
- mode?: "full" | "supplemental";
897
- };
898
- /**
899
- * Detect person names by looking up tokens against the
900
- * name corpus, then chaining adjacent name-like tokens.
901
- * Handles both Western and non-Western name patterns.
902
- *
903
- * Requires initNameCorpus() to have been called first.
904
- * If not initialized, returns an empty array.
905
- *
906
- * Scoring (Western):
907
- * TITLE + NAME/SURNAME → 0.95
908
- * NAME + NAME/SURNAME → 0.9
909
- * SURNAME + NAME/SURNAME → 0.9
910
- * NAME + CAPITALIZED → 0.7
911
- * ABBREVIATION + NAME → 0.7
912
- * Standalone NAME → 0.5 (low confidence)
913
- * Standalone SURNAME → skip (too ambiguous)
914
- *
915
- * Scoring (non-Western, when chain contains nonWestern tokens):
916
- * TITLE + (nonWestern|CAPITALIZED) → 0.95
917
- * JA_SUFFIX + (CAPITALIZED|nonWestern) → 0.9
918
- * ARABIC_CONNECTOR + nonWestern → 0.9
919
- * 2+ nonWestern tokens → 0.9
920
- * nonWestern + (CAPITALIZED|ABBREVIATION) → 0.9
921
- * Standalone nonWestern mid-sentence → 0.5
922
- */
923
- declare const detectNameCorpus: (fullText: string, ctx?: PipelineContext, options?: NameCorpusDetectionOptions) => Entity[];
924
- //#endregion
925
- //#region src/unified-search.d.ts
926
- type UnifiedResult = {
927
- /** All matches from both instances combined. */regexMatches: Match[];
928
- customRegexMatches: Match[];
929
- literalMatches: Match[];
930
- };
931
- declare const runUnifiedSearch: (instance: UnifiedSearchInstance, fullText: string) => UnifiedResult;
932
- //#endregion
933
- //#region src/regions.d.ts
934
- /**
935
- * Geographic regions and country code mappings for
936
- * scoping deny list dictionaries.
937
- */
938
- declare const REGIONS: {
939
- readonly Global: null;
940
- readonly International: null;
941
- readonly Europe: readonly ["AL", "AD", "AT", "BE", "BA", "BG", "HR", "CY", "CZ", "DK", "EE", "FI", "FR", "DE", "GR", "HU", "IS", "IE", "IT", "XK", "LV", "LI", "LT", "LU", "MD", "ME", "MK", "MT", "MC", "NL", "NO", "PL", "PT", "RO", "RS", "SK", "SI", "ES", "SE", "CH", "UA", "GB"];
942
- readonly Americas: readonly ["US", "CA", "MX", "BR", "AR", "CL", "CO", "PE", "EC", "VE", "UY", "PY", "BO", "CR", "PA", "DO", "GT", "HN", "SV", "NI", "CU"];
943
- readonly AsiaPacific: readonly ["AU", "NZ", "JP", "KR", "CN", "TW", "SG", "MY", "TH", "VN", "PH", "ID", "IN", "PK", "BD", "LK", "NP", "HK", "MO"];
944
- readonly MENA: readonly ["AE", "SA", "IL", "TR", "EG", "JO", "LB", "IQ", "IR", "QA", "KW", "BH", "OM", "MA", "TN", "DZ", "LY", "SY", "YE", "PS"];
945
- readonly SubSaharanAfrica: readonly ["ZA", "NG", "KE", "GH", "TZ", "ET", "SN", "CI", "CM", "UG", "RW", "MZ", "AO", "ZW", "BW", "NA", "MU"];
946
- readonly EU: readonly ["AT", "BE", "BG", "HR", "CY", "CZ", "DK", "EE", "FI", "FR", "DE", "GR", "HU", "IE", "IT", "LV", "LT", "LU", "MT", "NL", "PL", "PT", "RO", "SK", "SI", "ES", "SE"];
947
- readonly DACH: readonly ["DE", "AT", "CH"];
948
- readonly Nordics: readonly ["DK", "SE", "NO", "FI", "IS"];
949
- readonly CEE: readonly ["CZ", "SK", "PL", "HU", "RO", "BG", "HR", "SI", "LT", "LV", "EE"];
950
- readonly Anglosphere: readonly ["GB", "US", "CA", "AU", "NZ", "IE"];
951
- readonly Benelux: readonly ["BE", "NL", "LU"];
952
- readonly GulfStates: readonly ["AE", "SA", "QA", "KW", "BH", "OM"];
953
- readonly SouthAsia: readonly ["IN", "PK", "BD", "LK", "NP"];
954
- readonly EastAsia: readonly ["CN", "JP", "KR", "TW"];
955
- readonly SoutheastAsia: readonly ["SG", "MY", "TH", "VN", "PH", "ID"];
956
- readonly Oceania: readonly ["AU", "NZ"];
957
- };
958
- type RegionId = keyof typeof REGIONS;
959
- type RegionArrays = { [K in RegionId]: (typeof REGIONS)[K] };
960
- type NonNullRegion = { [K in RegionId as RegionArrays[K] extends null ? never : K]: RegionArrays[K] };
961
- type CountryCode = NonNullRegion[keyof NonNullRegion][number];
962
- /**
963
- * Expand region names to country codes and merge with
964
- * explicit country codes. Returns null when both inputs
965
- * are empty/undefined (meaning "match all countries").
966
- */
967
- declare const resolveCountries: (regions?: string[], countries?: string[]) => Set<string> | null;
968
- //#endregion
969
- //#region src/filters/false-positives.d.ts
970
- /** Ensure street-type vocabulary is loaded. */
971
- declare const initAddressComponents: () => Promise<void>;
972
- /**
973
- * Filter out entities that are likely false positives:
974
- * template placeholders, clause/section numbers,
975
- * standalone years, and generic legal role terms.
976
- *
977
- * Runs as a post-processing step after all detection
978
- * layers have merged.
979
- */
980
- declare const filterFalsePositives: (entities: Entity[], ctx?: PipelineContext, fullText?: string) => Entity[];
981
- //#endregion
982
- //#region src/filters/confidence-boost.d.ts
983
- /**
984
- * Boost confidence of near-miss NER entities that appear
985
- * near high-confidence detections (regex, trigger phrase).
986
- *
987
- * If an NER entity scored between (threshold - 0.15) and
988
- * threshold, count how many confirmed entities exist within
989
- * a 150-char window. Add +0.05 per co-located entity.
990
- * If the boosted score crosses the threshold, include it.
991
- *
992
- * Only mutates score on near-miss entities; high-confidence
993
- * entities pass through unchanged.
994
- */
995
- declare const boostNearMissEntities: (entities: Entity[], threshold: number) => Entity[];
996
- //#endregion
997
- //#region src/filters/hotword-rules.d.ts
998
- type HotwordRule = {
999
- hotwords: string[];
1000
- targetLabels: string[];
1001
- scoreAdjustment: number;
1002
- reclassifyTo?: string;
1003
- proximityBefore: number;
1004
- proximityAfter: number;
1005
- };
1006
- /**
1007
- * Load hotword rules from the data package.
1008
- * Safe to call multiple times; subsequent calls
1009
- * are no-ops.
1010
- */
1011
- declare const initHotwordRules: () => Promise<void>;
1012
- /**
1013
- * Apply hotword context rules to detected entities.
1014
- *
1015
- * Scans `fullText` once with a single AC automaton
1016
- * for all hotwords across all rules, then checks
1017
- * proximity to each entity. Distance-decayed
1018
- * adjustment: closer hotwords give a stronger boost.
1019
- *
1020
- * Returns a new array; input entities are not mutated.
1021
- */
1022
- declare const applyHotwordRules: (entities: Entity[], fullText: string) => Entity[];
1023
- //#endregion
1024
- //#region src/filters/zone-classifier.d.ts
1025
- type DocumentZone = "header" | "signature" | "body" | "table";
1026
- type ZoneSpan = {
1027
- zone: DocumentZone;
1028
- start: number;
1029
- end: number;
1030
- };
1031
- /**
1032
- * Additive score adjustments per document zone.
1033
- * Header and signature blocks are dense with PII;
1034
- * tables often contain structured identifying data.
1035
- */
1036
- declare const ZONE_SCORE_ADJUSTMENTS: {
1037
- readonly header: 0.1;
1038
- readonly signature: 0.15;
1039
- readonly body: 0;
1040
- readonly table: 0.05;
1041
- };
1042
- /**
1043
- * Ensure config data is loaded. Call once before
1044
- * classifyZones. Safe to call multiple times.
1045
- */
1046
- declare const initZoneClassifier: (ctx?: PipelineContext) => Promise<void>;
1047
- /**
1048
- * Classify a document into zones based on
1049
- * structural heuristics. Zones are non-overlapping
1050
- * and cover the entire text.
1051
- *
1052
- * Must call `initZoneClassifier()` first.
1053
- */
1054
- declare const classifyZones: (fullText: string, ctx?: PipelineContext) => ZoneSpan[];
1055
- /**
1056
- * Apply zone-based score adjustments to entities.
1057
- * Entities in header/signature/table zones get a
1058
- * small additive boost reflecting the higher PII
1059
- * density in those regions.
1060
- *
1061
- * Returns a new array; does not mutate inputs.
1062
- */
1063
- declare const applyZoneAdjustments: (entities: Entity[], zones: ZoneSpan[]) => Entity[];
1064
- //#endregion
1065
- //#region src/gliner/types.d.ts
1066
- /**
1067
- * GLiNER inference types.
1068
- *
1069
- * Forked from gliner@0.0.19 (MIT), stripped to runtime-
1070
- * agnostic core. Original: github.com/Ingvarstep/GLiNER.js
1071
- */
1072
- type EntityResult = {
1073
- spanText: string;
1074
- start: number;
1075
- end: number;
1076
- label: string;
1077
- score: number;
1078
- };
1079
- /**
1080
- * Raw inference output: per-batch array of
1081
- * [spanText, start, end, label, score] tuples.
1082
- */
1083
- type RawInferenceResult = [string, number, number, string, number][][];
1084
- //#endregion
1085
- //#region src/gliner/decoder.d.ts
1086
- /**
1087
- * Decode span-level model logits into entity results.
1088
- */
1089
- declare const decodeSpans: (batchSize: number, inputLength: number, maxWidth: number, numEntities: number, texts: string[], batchIds: number[], batchWordsStartIdx: number[][], batchWordsEndIdx: number[][], idToClass: Record<number, string>, modelOutput: ArrayLike<number>, flatNer: boolean, threshold: number, multiLabel: boolean) => RawInferenceResult;
1090
- //#endregion
1091
- //#region src/gliner/token-decoder.d.ts
1092
- /**
1093
- * Decode token-level BIO logits into entity spans.
1094
- *
1095
- * For each word, checks if the B(egin) logit for any class
1096
- * exceeds the threshold. If so, extends the span by consuming
1097
- * subsequent I(nside) tokens of the same class.
1098
- */
1099
- declare const decodeTokenSpans: (batchSize: number, numWords: number, numEntities: number, texts: string[], batchIds: number[], batchWordsStartIdx: number[][], batchWordsEndIdx: number[][], idToClass: Record<number, string>, modelOutput: ArrayLike<number>, threshold: number) => RawInferenceResult;
1100
- //#endregion
1101
- //#region src/gliner/processor.d.ts
1102
- /** Tokenize text into words with character offsets. */
1103
- declare const tokenizeText: (text: string) => [words: string[], starts: number[], ends: number[]];
1104
- /** Prepare a complete batch for ONNX inference. */
1105
- declare const prepareBatch: (tokenizer: Tokenizer, texts: string[], entities: string[], maxWidth: number) => {
1106
- inputsIds: number[][];
1107
- attentionMasks: number[][];
1108
- wordsMasks: number[][];
1109
- textLengths: number[];
1110
- spanIdxs: number[][][];
1111
- spanMasks: boolean[][];
1112
- idToClass: Record<number, string>;
1113
- batchTokens: string[][];
1114
- batchWordsStartIdx: number[][];
1115
- batchWordsEndIdx: number[][];
1116
- };
1117
- //#endregion
1118
- //#region src/util/chunker.d.ts
1119
- /** A chunk paired with its start offset in the source text. */
1120
- type ChunkSpan = {
1121
- text: string;
1122
- offset: number;
1123
- };
1124
- /**
1125
- * Split text into overlapping chunks, each paired with its
1126
- * exact start offset in the source text.
1127
- *
1128
- * Carrying the offset out of the splitter is the robust way to
1129
- * map chunk-local entity offsets back to document offsets:
1130
- * downstream code never has to re-locate a chunk by content
1131
- * search (which mis-locates when boilerplate repeats; see
1132
- * computeChunkOffsets).
1133
- *
1134
- * Character-based splitting (rough token approximation for
1135
- * GLiNER's ~512 token window); breaks at sentence boundaries
1136
- * when possible.
1137
- */
1138
- declare const chunkTextWithOffsets: (text: string) => ChunkSpan[];
1139
- /**
1140
- * Split text into overlapping chunks for GLiNER's ~512 token
1141
- * context window. Character-based splitting (rough token
1142
- * approximation); breaks at sentence boundaries when possible.
1143
- *
1144
- * Prefer chunkTextWithOffsets when you also need each chunk's
1145
- * document offset.
1146
- */
1147
- declare const chunkText: (text: string) => string[];
1148
- /**
1149
- * Compute the start offset of each chunk within the original
1150
- * document text by content search.
1151
- *
1152
- * @deprecated Re-locates each chunk with `indexOf`, which can
1153
- * match the wrong position when identical content repeats in
1154
- * the document (common in boilerplate-heavy legal text) and
1155
- * then desyncs every subsequent offset. Use
1156
- * `chunkTextWithOffsets`, which carries exact offsets out of
1157
- * the splitter.
1158
- */
1159
- declare const computeChunkOffsets: (fullText: string, chunks: string[]) => number[];
1160
- /**
1161
- * Merge entities from overlapping chunks back to
1162
- * document-level offsets. Deduplicates entities that
1163
- * appear in overlap regions (keeps highest score).
1164
- *
1165
- * Dedup invariant: each incoming entity is compared
1166
- * against the highest-scored same-label near-dup in
1167
- * its proximity window. If it loses, it is dropped.
1168
- * This does NOT guarantee that all pairwise near-dup
1169
- * relationships in the output are resolved; a lower-
1170
- * scored entity can survive if the bridging entity
1171
- * that would have replaced it was itself dropped by
1172
- * a higher-scored match.
1173
- *
1174
- * Uses a reverse-scan over the sorted merged array
1175
- * so each entity only compares against nearby
1176
- * predecessors — O(n * w) average where w is the max
1177
- * entities per POSITION_THRESHOLD window, O(n²) worst
1178
- * case when replacements dominate (splice is O(n)).
1179
- */
1180
- declare const mergeChunkEntities: (chunkOffsets: number[], chunkResults: Entity[][]) => Entity[];
1181
- //#endregion
1182
- //#region src/util/levenshtein.d.ts
1183
- /**
1184
- * Compute the Levenshtein edit distance between two
1185
- * strings. O(n*m) time, O(min(n,m)) space using a
1186
- * single-row DP approach.
1187
- */
1188
- declare const levenshtein: (rawA: string, rawB: string) => number;
1189
- //#endregion
1190
- //#region src/util/normalize.d.ts
1191
- /**
1192
- * Normalize typographic variants for search matching.
1193
- *
1194
- * Legal documents (especially Czech/German) use
1195
- * non-breaking spaces, smart quotes, and en/em dashes
1196
- * that differ from their ASCII equivalents. Since all
1197
- * replacements are same-length (single code unit →
1198
- * single code unit), character offsets remain valid.
1199
- *
1200
- * Lives here (application layer) rather than in the
1201
- * AC library: what to normalize is domain-specific.
1202
- *
1203
- * Uses a char-code lookup (`Map<number, number>`) and
1204
- * `Uint16Array` instead of 7 sequential `replaceAll`
1205
- * calls. For a 50 KB document this eliminates ~350 KB
1206
- * of intermediate string allocations.
1207
- *
1208
- * When no replaceable characters are present (common
1209
- * for plain-text inputs), a fast-path scan returns the
1210
- * original string without any allocation. When special
1211
- * characters exist, the string is scanned twice: once
1212
- * to detect, once to build the replacement array.
1213
- */
1214
- declare const normalizeForSearch: (text: string) => string;
1215
- //#endregion
1216
- export { type AnonymisationOperator, CURRENCY_PATTERN_META, type ChunkSpan, type CountryCode, type CustomDenyListEntry, type CustomRegexPattern, DATE_PATTERN_META, DEFAULT_ENTITY_LABELS, DEFAULT_OPERATOR_CONFIG, DETECTION_SOURCES, DETECTOR_PRIORITY, type DefinitionPattern, type DenyListCategory, type DenyListData, type DetectionSource, type Dictionaries, type DictionaryMeta, type DocumentZone, type Entity, type EntityResult, type GazetteerData, type GazetteerEntry, type HotwordRule, type NameCorpusData, type NerInferenceFn, OPERATOR_REGISTRY, OPERATOR_TYPES, type OperatorConfig, type OperatorType, type PipelineConfig, type PipelineContext, type PipelineOptions, type PipelineSearchOptions, REGEX_META, REGEX_PATTERNS, REGIONS, type RawInferenceResult, type RedactionResult, type RegexMeta, type RegionId, type ReviewDecision, type ReviewedEntity, type TriggerExtension, type TriggerGroupConfig, type TriggerRule, type TriggerStrategy, type TriggerValidation, type UnifiedResult, type UnifiedSearchInstance, ZONE_SCORE_ADJUSTMENTS, type ZoneSpan, applyHotwordRules, applyZoneAdjustments, boostNearMissEntities, buildDenyList, buildGazetteerPatterns, buildLegalFormPatterns, buildPlaceholderMap, buildStreetTypePatterns, buildTriggerPatterns, buildUnifiedSearch, chunkText, chunkTextWithOffsets, classifyZones, computeChunkOffsets, corefKey, createPipelineContext, deanonymise, decodeSpans, decodeTokenSpans, detectNameCorpus, ensureDenyListData, exportRedactionKey, extractDefinedTerms, filterFalsePositives, findCoreferenceSpans, getCurrencyPatterns, getDatePatterns, getNameCorpusNonWesternNames, initAddressComponents, initHotwordRules, initNameCorpus, initZoneClassifier, levenshtein, mergeAndDedup, mergeChunkEntities, normalizeForSearch, prepareBatch, preparePipelineSearch, processAddressSeeds, processDenyListMatches, processGazetteerMatches, processLegalFormMatches, processRegexMatches, processTriggerMatches, propagateOrgNames, redactText, resolveCountries, resolveOperator, runPipeline, runUnifiedSearch, sanitizeEntities, tokenizeText, warmLegalRoleHeads };
18
+ export { type AnonymisationOperator, type CustomDenyListEntry, type CustomRegexPattern, DEFAULT_ENTITY_LABELS, DEFAULT_NATIVE_PIPELINE_CONFIG, DEFAULT_NATIVE_PIPELINE_WARMUPS, DETECTION_SOURCES, DETECTOR_PRIORITY, DefaultNativePipelinePackageFileOptions, DefaultNativePipelinePackageOptions, DefaultNativePipelineWarmup, type DenyListCategory, type DetectionSource, type Dictionaries, type DictionaryMeta, type Entity, type GazetteerEntry, LoadNativeBindingOptions, NativeAnonymizeBinding, NativeAnonymizerFromConfigOptions, NativeAnonymizerFromPackageOptions, NativeBindingVersionOptions, NativeDiagnosticsBatchCallback, NativeLibc, NativeNormalizeOptions, NativeOperatorConfig, type NativePipelineBuildOptions, type NativePipelineCompatibility, NativePipelineEntity, NativePipelineFromPackageOptions, NativePipelinePackageFileOptions, type NativePipelinePackageOptions, type NativePipelineUnsupportedFeature, NativePreparedSearchBinding, type NativePreparedSearchConfig, NativeRedactionResult, NativeRequire, NativeResultEventCallback, NativeSdkOptions, NativeSdkPackageOptions, NativeSearchPackageInput, NativeSearchPackageOptions, NativeStaticRedactionResult, OPERATOR_TYPES, type OperatorConfig, type OperatorType, type PipelineConfig, PreparedAnonymizer, PreparedNativeAnonymizer, PreparedNativePipeline, PreparedSearch, type RedactionResult, type ReviewDecision, type ReviewedEntity, SharedNativeDiagnosticsJsonOptions, SharedNativeDiagnosticsStreamJsonOptions, SharedNativePreparedPackageOptions, SharedNativeRedactTextJsonOptions, SharedNativeRedactTextOptions, SharedNativeRedactTextStreamJsonOptions, SharedNativeSearchPackageOptions, type TriggerExtension, type TriggerGroupConfig, type TriggerRule, type TriggerStrategy, type TriggerValidation, assertNativeBindingVersion, assertNativePipelineSupported, availableDefaultNativePipelineLanguages, available_default_native_pipeline_languages, createNativeAnonymizerFromConfig, createNativeAnonymizerFromPackage, createNativePipelineFromConfig, createNativePipelineFromDefaultPackage, createNativePipelineFromPackage, createNativePipelineFromPackageFile, create_native_pipeline_from_default_package, deanonymise, diagnostics_json, diagnostics_stream_json, encodeNativeSearchConfig, encodeNativeSearchConfigInput, exportRedactionKey, getDefaultNativePipeline, getNativeBindingVersion, getNativePipelineCompatibility, get_default_native_pipeline, loadNativeAnonymizeBinding, load_prepared_package, load_prepared_package_file, native_package_version, normalize_for_search, preloadDefaultNativePipeline, preloadDefaultNativePipelineAsync, preload_default_native_pipeline, prepareNativePipelineConfig, prepareNativePipelinePackage, prepareNativeSearchPackage, prepare_search_package, readDefaultNativePipelinePackageFile, readDefaultNativePipelinePackageFileAsync, readNativePipelinePackageFile, readNativePipelinePackageFileAsync, read_default_native_pipeline_package_file, redactDefaultText, redactDefaultTextJson, redact_default_text, redact_default_text_json, redact_text, redact_text_json, redact_text_stream_json, summary_diagnostics_json };
1217
19
  //# sourceMappingURL=index.d.mts.map