@stll/anonymize 2.0.0-alpha.1 → 2.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/index.d.mts +2 -2
- package/dist/index.mjs +1 -1
- package/dist/native-node.d.mts +18 -4
- package/dist/native-node.mjs +1 -1
- package/dist/native-node2.d.mts +2 -2
- package/dist/native-node2.mjs +106 -12731
- package/dist/native-node2.mjs.map +1 -1
- package/dist/native.d.mts +526 -714
- package/dist/native.mjs.map +1 -1
- package/dist/native2.d.mts +2 -2
- package/package.json +27 -13
- package/dist/address-boundaries.mjs +0 -197
- package/dist/address-boundaries.mjs.map +0 -1
- package/dist/address-jurisdiction-prefixes.mjs +0 -16
- package/dist/address-jurisdiction-prefixes.mjs.map +0 -1
- package/dist/address-stop-keywords.mjs +0 -148
- package/dist/address-stop-keywords.mjs.map +0 -1
- package/dist/address-stopwords.mjs +0 -84
- package/dist/address-stopwords.mjs.map +0 -1
- package/dist/address-unit-abbreviations.mjs +0 -15
- package/dist/address-unit-abbreviations.mjs.map +0 -1
- package/dist/allow-list.mjs +0 -196
- package/dist/allow-list.mjs.map +0 -1
- package/dist/clause-noun-heads.mjs +0 -79
- package/dist/clause-noun-heads.mjs.map +0 -1
- package/dist/common-words-en.mjs +0 -9887
- package/dist/common-words-en.mjs.map +0 -1
- package/dist/coreference-org-determiners.mjs +0 -19
- package/dist/coreference-org-determiners.mjs.map +0 -1
- package/dist/coreference.cs.mjs +0 -14
- package/dist/coreference.cs.mjs.map +0 -1
- package/dist/coreference.de.mjs +0 -14
- package/dist/coreference.de.mjs.map +0 -1
- package/dist/coreference.en.mjs +0 -14
- package/dist/coreference.en.mjs.map +0 -1
- package/dist/coreference.es.mjs +0 -22
- package/dist/coreference.es.mjs.map +0 -1
- package/dist/coreference.fr.mjs +0 -32
- package/dist/coreference.fr.mjs.map +0 -1
- package/dist/coreference.it.mjs +0 -27
- package/dist/coreference.it.mjs.map +0 -1
- package/dist/coreference.pl.mjs +0 -27
- package/dist/coreference.pl.mjs.map +0 -1
- package/dist/coreference.pt-br.mjs +0 -14
- package/dist/coreference.pt-br.mjs.map +0 -1
- package/dist/coreference.sk.mjs +0 -27
- package/dist/coreference.sk.mjs.map +0 -1
- package/dist/currencies.mjs +0 -231
- package/dist/currencies.mjs.map +0 -1
- package/dist/date-months.mjs +0 -618
- package/dist/date-months.mjs.map +0 -1
- package/dist/defined-term-heads.mjs +0 -15
- package/dist/defined-term-heads.mjs.map +0 -1
- package/dist/document-structure-headings.mjs +0 -90
- package/dist/document-structure-headings.mjs.map +0 -1
- package/dist/false-positive-shapes.mjs +0 -36
- package/dist/false-positive-shapes.mjs.map +0 -1
- package/dist/generic-roles.mjs +0 -244
- package/dist/generic-roles.mjs.map +0 -1
- package/dist/hotword-rules.mjs +0 -149
- package/dist/hotword-rules.mjs.map +0 -1
- package/dist/legal-form-leading-clauses.mjs +0 -23
- package/dist/legal-form-leading-clauses.mjs.map +0 -1
- package/dist/legal-forms.mjs +0 -2115
- package/dist/legal-forms.mjs.map +0 -1
- package/dist/legal-role-heads.cs.mjs +0 -48
- package/dist/legal-role-heads.cs.mjs.map +0 -1
- package/dist/legal-role-heads.de.mjs +0 -33
- package/dist/legal-role-heads.de.mjs.map +0 -1
- package/dist/legal-role-heads.en.mjs +0 -37
- package/dist/legal-role-heads.en.mjs.map +0 -1
- package/dist/legal-role-heads.es.mjs +0 -54
- package/dist/legal-role-heads.es.mjs.map +0 -1
- package/dist/legal-role-heads.fr.mjs +0 -72
- package/dist/legal-role-heads.fr.mjs.map +0 -1
- package/dist/legal-role-heads.it.mjs +0 -68
- package/dist/legal-role-heads.it.mjs.map +0 -1
- package/dist/legal-role-heads.pl.mjs +0 -84
- package/dist/legal-role-heads.pl.mjs.map +0 -1
- package/dist/legal-role-heads.pt-br.mjs +0 -63
- package/dist/legal-role-heads.pt-br.mjs.map +0 -1
- package/dist/legal-role-heads.sk.mjs +0 -80
- package/dist/legal-role-heads.sk.mjs.map +0 -1
- package/dist/manifest.mjs +0 -69
- package/dist/manifest.mjs.map +0 -1
- package/dist/names-exclusions.mjs +0 -223
- package/dist/names-exclusions.mjs.map +0 -1
- package/dist/names-first.mjs +0 -418
- package/dist/names-first.mjs.map +0 -1
- package/dist/names-nw-ar.mjs +0 -202
- package/dist/names-nw-ar.mjs.map +0 -1
- package/dist/names-nw-excluded-allcaps.mjs +0 -112
- package/dist/names-nw-excluded-allcaps.mjs.map +0 -1
- package/dist/names-nw-fil.mjs +0 -202
- package/dist/names-nw-fil.mjs.map +0 -1
- package/dist/names-nw-id.mjs +0 -210
- package/dist/names-nw-id.mjs.map +0 -1
- package/dist/names-nw-in.mjs +0 -526
- package/dist/names-nw-in.mjs.map +0 -1
- package/dist/names-nw-ja-latn.mjs +0 -260
- package/dist/names-nw-ja-latn.mjs.map +0 -1
- package/dist/names-nw-ko.mjs +0 -162
- package/dist/names-nw-ko.mjs.map +0 -1
- package/dist/names-nw-th.mjs +0 -188
- package/dist/names-nw-th.mjs.map +0 -1
- package/dist/names-nw-vi.mjs +0 -151
- package/dist/names-nw-vi.mjs.map +0 -1
- package/dist/names-nw-zh-latn.mjs +0 -197
- package/dist/names-nw-zh-latn.mjs.map +0 -1
- package/dist/names-surnames.mjs +0 -113
- package/dist/names-surnames.mjs.map +0 -1
- package/dist/names-title-tokens.mjs +0 -40
- package/dist/names-title-tokens.mjs.map +0 -1
- package/dist/organization-unit-heads.mjs +0 -20
- package/dist/organization-unit-heads.mjs.map +0 -1
- package/dist/person-stopwords.mjs +0 -211
- package/dist/person-stopwords.mjs.map +0 -1
- package/dist/section-headings.mjs +0 -64
- package/dist/section-headings.mjs.map +0 -1
- package/dist/sentence-verb-indicators.mjs +0 -232
- package/dist/sentence-verb-indicators.mjs.map +0 -1
- package/dist/signing-clauses.mjs +0 -102
- package/dist/signing-clauses.mjs.map +0 -1
- package/dist/stopwords.mjs +0 -9915
- package/dist/stopwords.mjs.map +0 -1
- package/dist/structural-single-cap-prefixes.mjs +0 -99
- package/dist/structural-single-cap-prefixes.mjs.map +0 -1
- package/dist/triggers.cs.mjs +0 -569
- package/dist/triggers.cs.mjs.map +0 -1
- package/dist/triggers.de.mjs +0 -139
- package/dist/triggers.de.mjs.map +0 -1
- package/dist/triggers.en.mjs +0 -119
- package/dist/triggers.en.mjs.map +0 -1
- package/dist/triggers.es.mjs +0 -96
- package/dist/triggers.es.mjs.map +0 -1
- package/dist/triggers.fr.mjs +0 -275
- package/dist/triggers.fr.mjs.map +0 -1
- package/dist/triggers.global.mjs +0 -79
- package/dist/triggers.global.mjs.map +0 -1
- package/dist/triggers.hu.mjs +0 -41
- package/dist/triggers.hu.mjs.map +0 -1
- package/dist/triggers.it.mjs +0 -74
- package/dist/triggers.it.mjs.map +0 -1
- package/dist/triggers.pl.mjs +0 -271
- package/dist/triggers.pl.mjs.map +0 -1
- package/dist/triggers.pt-br.mjs +0 -193
- package/dist/triggers.pt-br.mjs.map +0 -1
- package/dist/triggers.ro.mjs +0 -59
- package/dist/triggers.ro.mjs.map +0 -1
- package/dist/triggers.sk.mjs +0 -555
- package/dist/triggers.sk.mjs.map +0 -1
- package/dist/triggers.sv.mjs +0 -58
- package/dist/triggers.sv.mjs.map +0 -1
- package/dist/year-words.mjs +0 -62
- package/dist/year-words.mjs.map +0 -1
package/dist/native.d.mts
CHANGED
|
@@ -1,93 +1,95 @@
|
|
|
1
1
|
import { i as DetectionSource, n as DETECTION_SOURCES, o as OperatorType } from "./constants2.mjs";
|
|
2
|
-
import { Validator } from "@stll/stdnum";
|
|
3
|
-
import { TextSearch } from "@stll/text-search";
|
|
4
2
|
|
|
5
|
-
//#region src/
|
|
3
|
+
//#region src/native-search-config.d.ts
|
|
6
4
|
/**
|
|
7
|
-
*
|
|
5
|
+
* Structural type for the prepared static-search config the native binding
|
|
6
|
+
* consumes and the Rust assembler emits (`assembleStaticSearchConfigJson`).
|
|
7
|
+
*
|
|
8
|
+
* This config used to be built in TypeScript by `build-unified-search.ts`; that
|
|
9
|
+
* layer was retired in favor of the Rust assembler
|
|
10
|
+
* (`crates/anonymize-adapter-contract` `assemble_static_search_config`). The
|
|
11
|
+
* type now lives here as a pure, dependency-free description of the JSON the
|
|
12
|
+
* binding accepts on its `fromConfigJsonBytes` / prepare paths, so callers that
|
|
13
|
+
* hold a pre-assembled config keep a precise type without pulling in the
|
|
14
|
+
* deleted detector modules.
|
|
8
15
|
*/
|
|
9
|
-
type
|
|
16
|
+
type PatternSlice = {
|
|
10
17
|
start: number;
|
|
11
18
|
end: number;
|
|
19
|
+
};
|
|
20
|
+
type NativeSearchPatternKind = "literal" | "literal-with-options" | "regex" | "fuzzy";
|
|
21
|
+
type NativeSearchPattern = {
|
|
22
|
+
kind: NativeSearchPatternKind;
|
|
23
|
+
pattern: string;
|
|
24
|
+
distance?: number;
|
|
25
|
+
case_insensitive?: boolean;
|
|
26
|
+
whole_words?: boolean;
|
|
27
|
+
lazy?: boolean;
|
|
28
|
+
prefilter_any?: string[];
|
|
29
|
+
prefilter_case_insensitive?: boolean;
|
|
30
|
+
prefilter_regex?: string;
|
|
31
|
+
prefilter_window_bytes?: number;
|
|
32
|
+
prepared_artifact_policy?: "include" | "omit";
|
|
33
|
+
};
|
|
34
|
+
type NativeSearchOptions = {
|
|
35
|
+
literal_case_insensitive?: boolean;
|
|
36
|
+
literal_whole_words?: boolean;
|
|
37
|
+
regex_whole_words?: boolean;
|
|
38
|
+
regex_overlap_all?: boolean;
|
|
39
|
+
regex_artifact_policy?: "include" | "omit";
|
|
40
|
+
fuzzy_case_insensitive?: boolean;
|
|
41
|
+
fuzzy_whole_words?: boolean;
|
|
42
|
+
fuzzy_normalize_diacritics?: boolean;
|
|
43
|
+
};
|
|
44
|
+
type NativeRegexMatchMeta = {
|
|
12
45
|
label: string;
|
|
13
|
-
text: string;
|
|
14
46
|
score: number;
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
*/
|
|
21
|
-
type DetectedEntity = EntityBase & {
|
|
22
|
-
source: Exclude<DetectionSource, typeof DETECTION_SOURCES.COREFERENCE>;
|
|
47
|
+
source_detail?: string;
|
|
48
|
+
requires_validation?: boolean;
|
|
49
|
+
validator_id?: string;
|
|
50
|
+
validator_input?: string;
|
|
51
|
+
min_byte_length?: number;
|
|
23
52
|
};
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
* mention ("Acme" after "Acme Corp.").
|
|
28
|
-
*
|
|
29
|
-
* `corefSourceText` is required by construction, so an
|
|
30
|
-
* alias cannot exist without the link back to its source
|
|
31
|
-
* entity. Placeholder numbering reads it to give the
|
|
32
|
-
* alias the same placeholder as the source. The link
|
|
33
|
-
* travels with the entity instead of living in a
|
|
34
|
-
* side-channel map that a producer could forget to
|
|
35
|
-
* write — or that a later pass could clear.
|
|
36
|
-
*/
|
|
37
|
-
type CorefAliasEntity = EntityBase & {
|
|
38
|
-
source: typeof DETECTION_SOURCES.COREFERENCE; /** Full text of the source entity this alias refers to. */
|
|
39
|
-
corefSourceText: string;
|
|
53
|
+
type NativeSigningPlaceGuardData = {
|
|
54
|
+
prefix_phrases: string[];
|
|
55
|
+
suffix_phrases: string[];
|
|
40
56
|
};
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
57
|
+
type NativeDenyListFilterData = {
|
|
58
|
+
stopwords: string[];
|
|
59
|
+
allow_list: string[];
|
|
60
|
+
person_stopwords: string[];
|
|
61
|
+
person_trailing_nouns: string[];
|
|
62
|
+
address_stopwords: string[];
|
|
63
|
+
address_jurisdiction_prefixes: string[];
|
|
64
|
+
street_types: string[];
|
|
65
|
+
address_component_terms: string[];
|
|
66
|
+
ambiguous_street_type_terms: string[];
|
|
67
|
+
first_names: string[];
|
|
68
|
+
generic_roles: string[];
|
|
69
|
+
number_abbrev_prefixes: string[];
|
|
70
|
+
sentence_starters: string[];
|
|
71
|
+
trailing_address_word_exclusions: string[];
|
|
72
|
+
document_heading_words: string[];
|
|
73
|
+
document_heading_ordinal_markers: string[];
|
|
74
|
+
defined_term_cues: string[];
|
|
75
|
+
signing_place_guards: NativeSigningPlaceGuardData[];
|
|
54
76
|
};
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
source: "manual" | "confirmed-from-model";
|
|
77
|
+
type NativeDenyListMatchData = {
|
|
78
|
+
labels?: string[][];
|
|
79
|
+
label_table?: string[];
|
|
80
|
+
label_indices?: number[][];
|
|
81
|
+
custom_labels?: string[][];
|
|
82
|
+
custom_label_indices?: number[][];
|
|
83
|
+
originals: string[];
|
|
84
|
+
sources?: string[][];
|
|
85
|
+
source_table?: string[];
|
|
86
|
+
source_indices?: number[][];
|
|
87
|
+
filters?: NativeDenyListFilterData;
|
|
67
88
|
};
|
|
68
|
-
|
|
69
|
-
type TriggerStrategy = {
|
|
89
|
+
type NativeTriggerStrategy = {
|
|
70
90
|
type: "to-next-comma";
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
* the value scan, in addition to commas/newlines. Useful
|
|
74
|
-
* for triggers like court names that may continue past
|
|
75
|
-
* a missing comma into adjacent clause text ("Městským
|
|
76
|
-
* soudem v Praze dne 1. 1. 2020"); listing `"dne"` here
|
|
77
|
-
* stops the scan at the date boundary. Matched on a
|
|
78
|
-
* word-boundary, case-insensitive.
|
|
79
|
-
*/
|
|
80
|
-
stopWords?: string[];
|
|
81
|
-
/**
|
|
82
|
-
* Hard cap on the captured span length, in characters,
|
|
83
|
-
* regardless of where the next comma / stop char sits.
|
|
84
|
-
* Use for triggers that label short formulaic phrases
|
|
85
|
-
* ("State of Delaware") and must not absorb the rest
|
|
86
|
-
* of a long forum-selection clause when the comma is
|
|
87
|
-
* sentences away. Falls back to the default 100-char
|
|
88
|
-
* fallback when omitted.
|
|
89
|
-
*/
|
|
90
|
-
maxLength?: number;
|
|
91
|
+
stop_words?: string[];
|
|
92
|
+
max_length?: number;
|
|
91
93
|
} | {
|
|
92
94
|
type: "to-end-of-line";
|
|
93
95
|
} | {
|
|
@@ -97,23 +99,13 @@ type TriggerStrategy = {
|
|
|
97
99
|
type: "company-id-value";
|
|
98
100
|
} | {
|
|
99
101
|
type: "address";
|
|
100
|
-
|
|
102
|
+
max_chars?: number;
|
|
101
103
|
} | {
|
|
102
|
-
/**
|
|
103
|
-
* Extract the first regex match in the value text.
|
|
104
|
-
* Useful for shape-bounded values that follow a
|
|
105
|
-
* label on the same line as other fields, where
|
|
106
|
-
* `to-end-of-line` would over-capture. The pattern
|
|
107
|
-
* is anchored to the start of the (already
|
|
108
|
-
* leading-whitespace-stripped) value, so use
|
|
109
|
-
* `(?:.*?)` prefix only when intentional.
|
|
110
|
-
*/
|
|
111
104
|
type: "match-pattern";
|
|
112
105
|
pattern: string;
|
|
113
106
|
flags?: string;
|
|
114
107
|
};
|
|
115
|
-
|
|
116
|
-
type TriggerValidation = {
|
|
108
|
+
type NativeTriggerValidation = {
|
|
117
109
|
type: "starts-uppercase";
|
|
118
110
|
} | {
|
|
119
111
|
type: "min-length";
|
|
@@ -129,291 +121,52 @@ type TriggerValidation = {
|
|
|
129
121
|
type: "matches-pattern";
|
|
130
122
|
pattern: string;
|
|
131
123
|
flags?: string;
|
|
132
|
-
}
|
|
133
|
-
/**
|
|
134
|
-
* Run a named stdnum validator (checksum + length)
|
|
135
|
-
* against the captured value. Keeps the trigger
|
|
136
|
-
* path symmetrical with the formatted-regex
|
|
137
|
-
* detectors so e.g. `CPF nº 00000000000` does not
|
|
138
|
-
* survive as a tax-ID entity.
|
|
139
|
-
*/
|
|
140
|
-
| {
|
|
141
|
-
type: "valid-id";
|
|
142
|
-
validator: ValidIdValidator;
|
|
143
|
-
};
|
|
144
|
-
/** Built-in stdnum validators that can be referenced
|
|
145
|
-
* by `valid-id` validations. */
|
|
146
|
-
type ValidIdValidator = "br.cpf" | "br.cnpj" | "us.rtn";
|
|
147
|
-
/** Auto-generated trigger variants — closed set. */
|
|
148
|
-
type TriggerExtension = "add-colon" | "add-trailing-space" | "add-colon-space" | "normalize-spaces";
|
|
149
|
-
/** V2 trigger config entry (JSON shape). */
|
|
150
|
-
type TriggerGroupConfig = {
|
|
151
|
-
id?: string;
|
|
152
|
-
triggers: string[];
|
|
153
|
-
label: string;
|
|
154
|
-
strategy: TriggerStrategy;
|
|
155
|
-
extensions?: TriggerExtension[];
|
|
156
|
-
validations?: TriggerValidation[];
|
|
157
|
-
/** When true, include the trigger text in the
|
|
158
|
-
* entity span (e.g., court names). */
|
|
159
|
-
includeTrigger?: boolean;
|
|
160
|
-
};
|
|
161
|
-
/** Compiled validation with pre-built regex. */
|
|
162
|
-
type CompiledValidation = {
|
|
163
|
-
type: "starts-uppercase";
|
|
164
|
-
re: RegExp;
|
|
165
|
-
} | {
|
|
166
|
-
type: "min-length";
|
|
167
|
-
min: number;
|
|
168
|
-
} | {
|
|
169
|
-
type: "max-length";
|
|
170
|
-
max: number;
|
|
171
|
-
} | {
|
|
172
|
-
type: "no-digits";
|
|
173
|
-
re: RegExp;
|
|
174
|
-
} | {
|
|
175
|
-
type: "has-digits";
|
|
176
|
-
re: RegExp;
|
|
177
|
-
} | {
|
|
178
|
-
type: "matches-pattern";
|
|
179
|
-
re: RegExp;
|
|
180
124
|
} | {
|
|
181
125
|
type: "valid-id";
|
|
182
|
-
validator:
|
|
183
|
-
check: (value: string) => boolean;
|
|
126
|
+
validator: string;
|
|
184
127
|
};
|
|
185
|
-
|
|
186
|
-
* Runtime rule — one per trigger string after
|
|
187
|
-
* expansion. Fed to the Aho-Corasick automaton.
|
|
188
|
-
*/
|
|
189
|
-
type TriggerRule = {
|
|
128
|
+
type NativeTriggerRule = {
|
|
190
129
|
trigger: string;
|
|
191
130
|
label: string;
|
|
192
|
-
strategy:
|
|
193
|
-
validations:
|
|
194
|
-
|
|
195
|
-
};
|
|
196
|
-
/** Per-label operator selection. Key is the entity label. */
|
|
197
|
-
type OperatorConfig = {
|
|
198
|
-
/** Operator per label. Missing labels default to "replace". */operators: Record<string, OperatorType>; /** Custom replacement string for the redact operator. */
|
|
199
|
-
redactString: string;
|
|
131
|
+
strategy: NativeTriggerStrategy;
|
|
132
|
+
validations: NativeTriggerValidation[];
|
|
133
|
+
include_trigger: boolean;
|
|
200
134
|
};
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
apply: (text: string, label: string, placeholder: string, redactString: string) => string;
|
|
135
|
+
type NativeTriggerData = {
|
|
136
|
+
rules: NativeTriggerRule[];
|
|
137
|
+
address_stop_keywords: string[];
|
|
138
|
+
party_position_terms: string[];
|
|
139
|
+
post_nominals: string[];
|
|
140
|
+
sentence_terminal_currency_terms: string[];
|
|
141
|
+
phone_extension_labels: string[];
|
|
142
|
+
number_markers: string[];
|
|
143
|
+
number_labels: string[];
|
|
211
144
|
};
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
/**
|
|
230
|
-
* Metadata for a single dictionary entry in the
|
|
231
|
-
* deny-list system. Mirrors the shape from
|
|
232
|
-
* the anonymize-data package so consumers can pass
|
|
233
|
-
* pre-loaded data without a runtime dependency.
|
|
234
|
-
*/
|
|
235
|
-
type DictionaryMeta = {
|
|
236
|
-
label: string;
|
|
237
|
-
category: DenyListCategory;
|
|
238
|
-
country: string | null;
|
|
239
|
-
};
|
|
240
|
-
/**
|
|
241
|
-
* Caller-supplied exact terms for deny-list matching.
|
|
242
|
-
* These entries are merged with the published deny-list
|
|
243
|
-
* dictionaries when `enableDenyList` is enabled.
|
|
244
|
-
*/
|
|
245
|
-
type CustomDenyListEntry = {
|
|
246
|
-
value: string;
|
|
247
|
-
label: string;
|
|
248
|
-
variants?: readonly string[];
|
|
249
|
-
};
|
|
250
|
-
/**
|
|
251
|
-
* Caller-supplied regex detector. The pattern is passed
|
|
252
|
-
* to the underlying text-search regex engine, so use its
|
|
253
|
-
* supported regex syntax. Inline flags such as `(?i)` are
|
|
254
|
-
* accepted when supported by that engine.
|
|
255
|
-
*/
|
|
256
|
-
type CustomRegexPattern = {
|
|
257
|
-
pattern: string;
|
|
258
|
-
label: string;
|
|
259
|
-
score?: number;
|
|
260
|
-
preparedArtifactPolicy?: "include" | "omit";
|
|
261
|
-
};
|
|
262
|
-
/**
|
|
263
|
-
* Pre-loaded dictionary data for dependency injection.
|
|
264
|
-
* Consumers that want name/city/deny-list detection
|
|
265
|
-
* load dictionaries themselves (e.g. from the
|
|
266
|
-
* anonymize-data package) and pass them here; the
|
|
267
|
-
* anonymize package has zero cross-package imports.
|
|
268
|
-
*
|
|
269
|
-
* All fields are optional. When a field is absent,
|
|
270
|
-
* the corresponding detection path is skipped (same
|
|
271
|
-
* behavior as when no dictionaries are available).
|
|
272
|
-
*/
|
|
273
|
-
type Dictionaries = {
|
|
274
|
-
/**
|
|
275
|
-
* First names per language code (e.g., "cs", "de").
|
|
276
|
-
* Merged with legacy config names at init time.
|
|
277
|
-
*/
|
|
278
|
-
firstNames?: Readonly<Record<string, readonly string[]>>;
|
|
279
|
-
/**
|
|
280
|
-
* Surnames per language code.
|
|
281
|
-
* Merged with legacy config names at init time.
|
|
282
|
-
*/
|
|
283
|
-
surnames?: Readonly<Record<string, readonly string[]>>;
|
|
284
|
-
/**
|
|
285
|
-
* Non-Western name tokens per locale code
|
|
286
|
-
* (e.g., "in", "ar", "ja-latn", "ko", "zh-latn",
|
|
287
|
-
* "th", "vi", "fil", "id"). Merged with bundled
|
|
288
|
-
* names-nw-*.json data at init time.
|
|
289
|
-
*/
|
|
290
|
-
nonWesternNames?: Readonly<Record<string, readonly string[]>>;
|
|
291
|
-
/**
|
|
292
|
-
* Pre-loaded deny-list dictionaries keyed by
|
|
293
|
-
* dictionary ID (e.g., "courts/CZ", "banks/DE").
|
|
294
|
-
* Each value is the array of terms for that
|
|
295
|
-
* dictionary.
|
|
296
|
-
*/
|
|
297
|
-
denyList?: Readonly<Record<string, readonly string[]>>;
|
|
298
|
-
/**
|
|
299
|
-
* Metadata per dictionary ID. Required when
|
|
300
|
-
* `denyList` is provided so the pipeline knows
|
|
301
|
-
* labels, categories, and country filters.
|
|
302
|
-
*/
|
|
303
|
-
denyListMeta?: Readonly<Record<string, DictionaryMeta>>;
|
|
304
|
-
/**
|
|
305
|
-
* Pre-loaded city names, already merged across
|
|
306
|
-
* all desired countries.
|
|
307
|
-
*
|
|
308
|
-
* Prefer `citiesByCountry` when callers also pass
|
|
309
|
-
* `denyListCountries` / `denyListRegions`; merged
|
|
310
|
-
* city arrays cannot be scoped after injection.
|
|
311
|
-
*/
|
|
312
|
-
cities?: readonly string[];
|
|
313
|
-
/**
|
|
314
|
-
* Pre-loaded city names keyed by ISO 3166-1 alpha-2
|
|
315
|
-
* country code. When provided, the deny-list builder
|
|
316
|
-
* applies `denyListCountries` / `denyListRegions`
|
|
317
|
-
* before adding city patterns to the search automaton.
|
|
318
|
-
*/
|
|
319
|
-
citiesByCountry?: Readonly<Record<string, readonly string[]>>;
|
|
145
|
+
type NativeLegalFormData = {
|
|
146
|
+
suffixes: string[];
|
|
147
|
+
normalized_boundary_suffixes: string[];
|
|
148
|
+
normalized_in_name_words: string[];
|
|
149
|
+
normalized_suffix_words: string[];
|
|
150
|
+
role_heads: string[];
|
|
151
|
+
sentence_verb_indicators: string[];
|
|
152
|
+
clause_noun_heads: string[];
|
|
153
|
+
connector_prose_heads: string[];
|
|
154
|
+
structural_single_cap_prefixes: string[];
|
|
155
|
+
leading_clause_phrases: string[];
|
|
156
|
+
leading_clause_direct_prefixes: string[];
|
|
157
|
+
connector_words: string[];
|
|
158
|
+
and_connector_words: string[];
|
|
159
|
+
in_name_prepositions: string[];
|
|
160
|
+
company_suffix_words: string[];
|
|
161
|
+
comma_gated_direct_prefixes: string[];
|
|
320
162
|
};
|
|
321
|
-
type
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
* Expected content language codes. When present, these
|
|
327
|
-
* derive default dictionary scopes for name corpus and
|
|
328
|
-
* deny-list matching unless the lower-level scope fields
|
|
329
|
-
* below are set explicitly.
|
|
330
|
-
*/
|
|
331
|
-
languages?: string[];
|
|
332
|
-
/**
|
|
333
|
-
* Convenience form for single-language documents. Ignored
|
|
334
|
-
* when `languages` is also provided.
|
|
335
|
-
*/
|
|
336
|
-
language?: string;
|
|
337
|
-
/**
|
|
338
|
-
* Enables legal-form organization detection.
|
|
339
|
-
* Required for typed callers; legacy untyped
|
|
340
|
-
* callers that omit this field are treated as
|
|
341
|
-
* enabled at runtime for backward compatibility.
|
|
342
|
-
*/
|
|
343
|
-
enableLegalForms: boolean;
|
|
344
|
-
/**
|
|
345
|
-
* Enables first-name/surname/title corpus matching.
|
|
346
|
-
* When deny-list mode is enabled, this also controls
|
|
347
|
-
* whether name-corpus entries are injected into the
|
|
348
|
-
* deny-list search automaton.
|
|
349
|
-
*/
|
|
350
|
-
enableNameCorpus: boolean;
|
|
351
|
-
/**
|
|
352
|
-
* Optional language scope for first-name/surname
|
|
353
|
-
* dictionaries, using the keys present in
|
|
354
|
-
* `dictionaries.firstNames` / `dictionaries.surnames`
|
|
355
|
-
* (for example `["en", "de"]`). When omitted, all
|
|
356
|
-
* injected name languages are used for backward
|
|
357
|
-
* compatibility.
|
|
358
|
-
*/
|
|
359
|
-
nameCorpusLanguages?: string[];
|
|
360
|
-
enableDenyList: boolean;
|
|
361
|
-
denyListCountries?: string[];
|
|
362
|
-
denyListRegions?: string[];
|
|
363
|
-
denyListExcludeCategories?: string[];
|
|
364
|
-
/**
|
|
365
|
-
* Caller-owned exact terms to match through the
|
|
366
|
-
* deny-list layer. Requires `enableDenyList: true`.
|
|
367
|
-
*/
|
|
368
|
-
customDenyList?: readonly CustomDenyListEntry[];
|
|
369
|
-
/**
|
|
370
|
-
* Caller-owned regex detectors. Requires
|
|
371
|
-
* `enableRegex: true`.
|
|
372
|
-
*/
|
|
373
|
-
customRegexes?: readonly CustomRegexPattern[];
|
|
374
|
-
enableGazetteer: boolean;
|
|
375
|
-
/**
|
|
376
|
-
* Detect country names (ISO 3166-1 names, curated
|
|
377
|
-
* aliases, alpha-3 codes). Defaults to true. Names
|
|
378
|
-
* span all manifest languages plus widely-used
|
|
379
|
-
* additions (Dutch, Russian, Chinese, Arabic, etc.).
|
|
380
|
-
*/
|
|
381
|
-
enableCountries?: boolean;
|
|
382
|
-
enableNer: boolean;
|
|
383
|
-
enableConfidenceBoost: boolean;
|
|
384
|
-
enableCoreference: boolean;
|
|
385
|
-
enableZoneClassification?: boolean;
|
|
386
|
-
enableHotwordRules?: boolean;
|
|
387
|
-
/**
|
|
388
|
-
* Requested output labels. An empty array means
|
|
389
|
-
* "do not filter by label" for deterministic
|
|
390
|
-
* detectors; NER falls back to DEFAULT_ENTITY_LABELS.
|
|
391
|
-
*/
|
|
392
|
-
labels: string[];
|
|
393
|
-
workspaceId: string;
|
|
394
|
-
/**
|
|
395
|
-
* Pre-loaded dictionary data for name, deny-list,
|
|
396
|
-
* and city detection. When omitted, dictionary-based
|
|
397
|
-
* detection paths are skipped. Consumers load from
|
|
398
|
-
* the anonymize-data package and pass the data here.
|
|
399
|
-
*/
|
|
400
|
-
dictionaries?: Dictionaries;
|
|
163
|
+
type NativeDateMonthData = Record<string, string[]>;
|
|
164
|
+
type NativeYearWordData = Record<string, string[]>;
|
|
165
|
+
type NativeDateData = {
|
|
166
|
+
month_names_by_language: NativeDateMonthData;
|
|
167
|
+
year_words_by_language: NativeYearWordData;
|
|
401
168
|
};
|
|
402
|
-
|
|
403
|
-
//#region src/detectors/regex.d.ts
|
|
404
|
-
type RegexMeta = {
|
|
405
|
-
label: string;
|
|
406
|
-
score: number;
|
|
407
|
-
sourceDetail?: Entity["sourceDetail"];
|
|
408
|
-
minByteLength?: number; /** Post-match stdnum validator for confirmation. */
|
|
409
|
-
validator?: Validator;
|
|
410
|
-
validatorId?: string; /** Extract the identifier portion when context is part of the regex span. */
|
|
411
|
-
validatorInput?: (text: string) => string;
|
|
412
|
-
validatorInputKind?: "digits-only" | "crypto-wallet-candidate";
|
|
413
|
-
};
|
|
414
|
-
type DateMonthData = Record<string, string[]>;
|
|
415
|
-
type YearWordData = Record<string, string[]>;
|
|
416
|
-
type MonetaryData = {
|
|
169
|
+
type NativeMonetaryData = {
|
|
417
170
|
currencies: {
|
|
418
171
|
codes: string[];
|
|
419
172
|
symbols: string[];
|
|
@@ -434,337 +187,20 @@ type MonetaryData = {
|
|
|
434
187
|
}>;
|
|
435
188
|
};
|
|
436
189
|
};
|
|
437
|
-
|
|
438
|
-
//#region src/context.d.ts
|
|
439
|
-
/**
|
|
440
|
-
* Compiled RegExp pattern used for coreference
|
|
441
|
-
* definition extraction.
|
|
442
|
-
*/
|
|
443
|
-
type DefinitionPattern = {
|
|
444
|
-
pattern: RegExp;
|
|
445
|
-
};
|
|
446
|
-
/**
|
|
447
|
-
* Cached data for the name corpus detector.
|
|
448
|
-
* Populated by initNameCorpus; consumed by
|
|
449
|
-
* detectNameCorpus and deny-list AC integration.
|
|
450
|
-
*/
|
|
451
|
-
type NameCorpusData = {
|
|
452
|
-
firstNames: ReadonlySet<string>;
|
|
453
|
-
surnames: ReadonlySet<string>;
|
|
454
|
-
titleTokens: ReadonlySet<string>;
|
|
455
|
-
/** Abbreviation-style titles whose trailing dot is
|
|
456
|
-
* part of the title, not a sentence boundary.
|
|
457
|
-
* Contains the lowercase, dot-stripped form
|
|
458
|
-
* (e.g., "dr", "smt", "atty"). */
|
|
459
|
-
titleAbbreviations: ReadonlySet<string>;
|
|
460
|
-
excludedWords: ReadonlySet<string>;
|
|
461
|
-
/** Lowercased common English words. A name chain whose
|
|
462
|
-
* every token is a common word (e.g. "Loan Documents",
|
|
463
|
-
* where "Loan" coincides with a Vietnamese given name)
|
|
464
|
-
* is treated as a common-word phrase, not a person. */
|
|
465
|
-
commonWords: ReadonlySet<string>; /** Non-Western name tokens merged across all locales. */
|
|
466
|
-
nonWesternNames: ReadonlySet<string>; /** All-caps acronyms excluded from name detection. */
|
|
467
|
-
excludedAllCaps: ReadonlySet<string>; /** Raw arrays exposed for deny-list AC integration. */
|
|
468
|
-
firstNamesList: readonly string[];
|
|
469
|
-
surnamesList: readonly string[];
|
|
470
|
-
titlesList: readonly string[];
|
|
471
|
-
excludedList: readonly string[];
|
|
472
|
-
nonWesternNamesList: readonly string[];
|
|
473
|
-
excludedAllCapsList: readonly string[];
|
|
474
|
-
};
|
|
475
|
-
/**
|
|
476
|
-
* All cached state for a single pipeline run (or
|
|
477
|
-
* sequence of runs sharing the same config). Replacing
|
|
478
|
-
* module-level singletons with this object enables
|
|
479
|
-
* concurrent pipelines with different configs and
|
|
480
|
-
* simplifies testing.
|
|
481
|
-
*
|
|
482
|
-
* Each field starts null and is populated lazily on
|
|
483
|
-
* first use by the corresponding loader function.
|
|
484
|
-
*/
|
|
485
|
-
type PipelineContext = {
|
|
486
|
-
search: UnifiedSearchInstance | null;
|
|
487
|
-
searchKey: string;
|
|
488
|
-
searchPromise: Promise<UnifiedSearchInstance> | null;
|
|
489
|
-
nativePipelinePackage: Uint8Array | null;
|
|
490
|
-
nativePipelinePackageKey: string;
|
|
491
|
-
nativePipelinePackagePromise: Promise<Uint8Array> | null;
|
|
492
|
-
nameCorpus: NameCorpusData | null;
|
|
493
|
-
nameCorpusKey: string;
|
|
494
|
-
nameCorpusPromise: Promise<void> | null;
|
|
495
|
-
stopwords: ReadonlySet<string> | null;
|
|
496
|
-
stopwordsPromise: Promise<ReadonlySet<string>> | null;
|
|
497
|
-
allowList: ReadonlySet<string> | null;
|
|
498
|
-
allowListPromise: Promise<ReadonlySet<string>> | null;
|
|
499
|
-
personStopwords: ReadonlySet<string> | null;
|
|
500
|
-
personStopwordsPromise: Promise<ReadonlySet<string>> | null;
|
|
501
|
-
definedTermHeads: ReadonlySet<string> | null;
|
|
502
|
-
definedTermHeadsPromise: Promise<ReadonlySet<string>> | null;
|
|
503
|
-
addressStopwords: ReadonlySet<string> | null;
|
|
504
|
-
addressStopwordsPromise: Promise<ReadonlySet<string>> | null; /** First-name exclusions for stopword filtering. */
|
|
505
|
-
firstNameExclusions: ReadonlySet<string> | null;
|
|
506
|
-
firstNameExclusionCorpusLen: number;
|
|
507
|
-
genericRoles: ReadonlySet<string> | null;
|
|
508
|
-
genericRolesPromise: Promise<ReadonlySet<string>> | null;
|
|
509
|
-
corefPatterns: DefinitionPattern[] | null;
|
|
510
|
-
corefPatternsKey: string;
|
|
511
|
-
corefPatternsPromise: Promise<DefinitionPattern[]> | null;
|
|
512
|
-
corefLoadAttempted: boolean;
|
|
513
|
-
roleStopSet: ReadonlySet<string> | null;
|
|
514
|
-
roleStopSetPromise: Promise<ReadonlySet<string>> | null;
|
|
515
|
-
zoneHeadingPatterns: RegExp[] | null;
|
|
516
|
-
zoneSigningPatterns: RegExp[] | null;
|
|
517
|
-
zoneInitPromise: Promise<void> | null;
|
|
518
|
-
};
|
|
519
|
-
//#endregion
|
|
520
|
-
//#region src/detectors/deny-list.d.ts
|
|
521
|
-
type DenyListFilterData = {
|
|
522
|
-
stopwords: string[];
|
|
523
|
-
allowList: string[];
|
|
524
|
-
personStopwords: string[];
|
|
525
|
-
personTrailingNouns: string[];
|
|
526
|
-
addressStopwords: string[];
|
|
527
|
-
addressJurisdictionPrefixes: string[];
|
|
528
|
-
streetTypes: string[];
|
|
529
|
-
addressComponentTerms: string[];
|
|
530
|
-
ambiguousStreetTypeTerms: string[];
|
|
531
|
-
firstNames: string[];
|
|
532
|
-
genericRoles: string[];
|
|
533
|
-
numberAbbrevPrefixes: string[];
|
|
534
|
-
sentenceStarters: string[];
|
|
535
|
-
trailingAddressWordExclusions: string[];
|
|
536
|
-
documentHeadingWords: string[];
|
|
537
|
-
documentHeadingOrdinalMarkers: string[];
|
|
538
|
-
definedTermCues: string[];
|
|
539
|
-
signingPlaceGuards: DenyListSigningPlaceGuardData[];
|
|
540
|
-
};
|
|
541
|
-
type DenyListSigningPlaceGuardData = {
|
|
542
|
-
prefixPhrases: string[];
|
|
543
|
-
suffixPhrases: string[];
|
|
544
|
-
};
|
|
545
|
-
/**
|
|
546
|
-
* Source tag for each pattern in the automaton.
|
|
547
|
-
* "deny-list" = standard deny list entry
|
|
548
|
-
* "city" = city dictionary entry
|
|
549
|
-
* "custom-deny-list" = caller-owned exact term
|
|
550
|
-
* "first-name" = name corpus first name
|
|
551
|
-
* "surname" = name corpus surname
|
|
552
|
-
* "title" = academic/professional title
|
|
553
|
-
*/
|
|
554
|
-
type PatternSource = "deny-list" | "city" | "custom-deny-list" | "first-name" | "surname" | "title";
|
|
555
|
-
type PatternLabels = string | string[];
|
|
556
|
-
type PatternSources = PatternSource | PatternSource[];
|
|
557
|
-
/**
|
|
558
|
-
* Pre-built deny list data. Constructed once by
|
|
559
|
-
* `buildDenyList`, reused across `processDenyListMatches`
|
|
560
|
-
* calls. Contains PatternEntry[] for the unified builder
|
|
561
|
-
* plus parallel label/source arrays for post-processing.
|
|
562
|
-
*/
|
|
563
|
-
type DenyListData = {
|
|
564
|
-
/**
|
|
565
|
-
* Maps pattern index → entity labels (plural).
|
|
566
|
-
* Same pattern can have multiple labels when it
|
|
567
|
-
* appears in multiple dictionaries (e.g., "Denver"
|
|
568
|
-
* is both a person name and a city name).
|
|
569
|
-
*/
|
|
570
|
-
labels: PatternLabels[]; /** Maps pattern index → labels contributed by custom entries. */
|
|
571
|
-
customLabels: (PatternLabels | undefined)[]; /** Maps pattern index → original pattern text. */
|
|
572
|
-
originals: string[]; /** Maps pattern index → source types (plural). */
|
|
573
|
-
sources: PatternSources[];
|
|
574
|
-
filters: DenyListFilterData;
|
|
575
|
-
};
|
|
576
|
-
//#endregion
|
|
577
|
-
//#region src/detectors/address-seeds.d.ts
|
|
578
|
-
type AddressSeedData = {
|
|
190
|
+
type NativeAddressSeedData = {
|
|
579
191
|
boundary_words: string[];
|
|
580
192
|
br_cep_cue_words: string[];
|
|
581
193
|
unit_abbreviations: string[];
|
|
582
194
|
};
|
|
583
|
-
|
|
584
|
-
//#region src/detectors/countries.d.ts
|
|
585
|
-
/**
|
|
586
|
-
* Pre-built country patterns + parallel label/source
|
|
587
|
-
* metadata. Constructed once and reused across pipeline
|
|
588
|
-
* runs.
|
|
589
|
-
*/
|
|
590
|
-
type CountryData = {
|
|
591
|
-
/** Maps local pattern index to entity label. Always "country". */labels: string[];
|
|
592
|
-
/**
|
|
593
|
-
* Maps local pattern index to the alpha-2 ISO code the
|
|
594
|
-
* pattern resolves to. Used for downstream coreference /
|
|
595
|
-
* placeholder grouping.
|
|
596
|
-
*/
|
|
597
|
-
isoCodes: string[]; /** Maps local pattern index to pattern variant kind. */
|
|
598
|
-
variants: CountryVariant[];
|
|
599
|
-
};
|
|
600
|
-
type CountryVariant = "name" | "alias" | "alpha3" | "alpha2";
|
|
601
|
-
//#endregion
|
|
602
|
-
//#region src/filters/confidence-boost.d.ts
|
|
603
|
-
type AddressContextData = {
|
|
195
|
+
type NativeAddressContextData = {
|
|
604
196
|
address_prepositions: string[];
|
|
605
197
|
temporal_prepositions: string[];
|
|
606
198
|
street_abbreviations: string[];
|
|
607
199
|
bare_house_stopwords: string[];
|
|
608
200
|
};
|
|
609
|
-
|
|
610
|
-
//#region src/build-unified-search.d.ts
|
|
611
|
-
type PatternSlice = {
|
|
612
|
-
start: number;
|
|
613
|
-
end: number;
|
|
614
|
-
};
|
|
615
|
-
type NativeSearchPatternKind = "literal" | "literal-with-options" | "regex" | "fuzzy";
|
|
616
|
-
type NativeSearchPattern = {
|
|
617
|
-
kind: NativeSearchPatternKind;
|
|
201
|
+
type NativeCoreferencePatternData = {
|
|
618
202
|
pattern: string;
|
|
619
|
-
|
|
620
|
-
case_insensitive?: boolean;
|
|
621
|
-
whole_words?: boolean;
|
|
622
|
-
lazy?: boolean;
|
|
623
|
-
prefilter_any?: string[];
|
|
624
|
-
prefilter_case_insensitive?: boolean;
|
|
625
|
-
prefilter_regex?: string;
|
|
626
|
-
prefilter_window_bytes?: number;
|
|
627
|
-
prepared_artifact_policy?: "include" | "omit";
|
|
628
|
-
};
|
|
629
|
-
type NativeSearchOptions = {
|
|
630
|
-
literal_case_insensitive?: boolean;
|
|
631
|
-
literal_whole_words?: boolean;
|
|
632
|
-
regex_whole_words?: boolean;
|
|
633
|
-
regex_overlap_all?: boolean;
|
|
634
|
-
regex_artifact_policy?: "include" | "omit";
|
|
635
|
-
fuzzy_case_insensitive?: boolean;
|
|
636
|
-
fuzzy_whole_words?: boolean;
|
|
637
|
-
fuzzy_normalize_diacritics?: boolean;
|
|
638
|
-
};
|
|
639
|
-
type NativeRegexMatchMeta = {
|
|
640
|
-
label: string;
|
|
641
|
-
score: number;
|
|
642
|
-
source_detail?: string;
|
|
643
|
-
requires_validation?: boolean;
|
|
644
|
-
validator_id?: string;
|
|
645
|
-
validator_input?: string;
|
|
646
|
-
min_byte_length?: number;
|
|
647
|
-
};
|
|
648
|
-
type NativeDenyListFilterData = {
|
|
649
|
-
stopwords: string[];
|
|
650
|
-
allow_list: string[];
|
|
651
|
-
person_stopwords: string[];
|
|
652
|
-
person_trailing_nouns: string[];
|
|
653
|
-
address_stopwords: string[];
|
|
654
|
-
address_jurisdiction_prefixes: string[];
|
|
655
|
-
street_types: string[];
|
|
656
|
-
address_component_terms: string[];
|
|
657
|
-
ambiguous_street_type_terms: string[];
|
|
658
|
-
first_names: string[];
|
|
659
|
-
generic_roles: string[];
|
|
660
|
-
number_abbrev_prefixes: string[];
|
|
661
|
-
sentence_starters: string[];
|
|
662
|
-
trailing_address_word_exclusions: string[];
|
|
663
|
-
document_heading_words: string[];
|
|
664
|
-
document_heading_ordinal_markers: string[];
|
|
665
|
-
defined_term_cues: string[];
|
|
666
|
-
signing_place_guards: NativeSigningPlaceGuardData[];
|
|
667
|
-
};
|
|
668
|
-
type NativeSigningPlaceGuardData = {
|
|
669
|
-
prefix_phrases: string[];
|
|
670
|
-
suffix_phrases: string[];
|
|
671
|
-
};
|
|
672
|
-
type NativeDenyListMatchData = {
|
|
673
|
-
labels?: string[][];
|
|
674
|
-
label_table?: string[];
|
|
675
|
-
label_indices?: number[][];
|
|
676
|
-
custom_labels?: string[][];
|
|
677
|
-
custom_label_indices?: number[][];
|
|
678
|
-
originals: string[];
|
|
679
|
-
sources?: string[][];
|
|
680
|
-
source_table?: string[];
|
|
681
|
-
source_indices?: number[][];
|
|
682
|
-
filters?: NativeDenyListFilterData;
|
|
683
|
-
};
|
|
684
|
-
type NativeTriggerStrategy = {
|
|
685
|
-
type: "to-next-comma";
|
|
686
|
-
stop_words?: string[];
|
|
687
|
-
max_length?: number;
|
|
688
|
-
} | {
|
|
689
|
-
type: "to-end-of-line";
|
|
690
|
-
} | {
|
|
691
|
-
type: "n-words";
|
|
692
|
-
count: number;
|
|
693
|
-
} | {
|
|
694
|
-
type: "company-id-value";
|
|
695
|
-
} | {
|
|
696
|
-
type: "address";
|
|
697
|
-
max_chars?: number;
|
|
698
|
-
} | {
|
|
699
|
-
type: "match-pattern";
|
|
700
|
-
pattern: string;
|
|
701
|
-
flags?: string;
|
|
702
|
-
};
|
|
703
|
-
type NativeTriggerValidation = {
|
|
704
|
-
type: "starts-uppercase";
|
|
705
|
-
} | {
|
|
706
|
-
type: "min-length";
|
|
707
|
-
min: number;
|
|
708
|
-
} | {
|
|
709
|
-
type: "max-length";
|
|
710
|
-
max: number;
|
|
711
|
-
} | {
|
|
712
|
-
type: "no-digits";
|
|
713
|
-
} | {
|
|
714
|
-
type: "has-digits";
|
|
715
|
-
} | {
|
|
716
|
-
type: "matches-pattern";
|
|
717
|
-
pattern: string;
|
|
718
|
-
flags?: string;
|
|
719
|
-
} | {
|
|
720
|
-
type: "valid-id";
|
|
721
|
-
validator: string;
|
|
722
|
-
};
|
|
723
|
-
type NativeTriggerRule = {
|
|
724
|
-
trigger: string;
|
|
725
|
-
label: string;
|
|
726
|
-
strategy: NativeTriggerStrategy;
|
|
727
|
-
validations: NativeTriggerValidation[];
|
|
728
|
-
include_trigger: boolean;
|
|
729
|
-
};
|
|
730
|
-
type NativeTriggerData = {
|
|
731
|
-
rules: NativeTriggerRule[];
|
|
732
|
-
address_stop_keywords: string[];
|
|
733
|
-
party_position_terms: string[];
|
|
734
|
-
post_nominals: string[];
|
|
735
|
-
sentence_terminal_currency_terms: string[];
|
|
736
|
-
phone_extension_labels: string[];
|
|
737
|
-
number_markers: string[];
|
|
738
|
-
number_labels: string[];
|
|
739
|
-
};
|
|
740
|
-
type NativeLegalFormData = {
|
|
741
|
-
suffixes: string[];
|
|
742
|
-
normalized_boundary_suffixes: string[];
|
|
743
|
-
normalized_in_name_words: string[];
|
|
744
|
-
normalized_suffix_words: string[];
|
|
745
|
-
role_heads: string[];
|
|
746
|
-
sentence_verb_indicators: string[];
|
|
747
|
-
clause_noun_heads: string[];
|
|
748
|
-
connector_prose_heads: string[];
|
|
749
|
-
structural_single_cap_prefixes: string[];
|
|
750
|
-
leading_clause_phrases: string[];
|
|
751
|
-
leading_clause_direct_prefixes: string[];
|
|
752
|
-
connector_words: string[];
|
|
753
|
-
and_connector_words: string[];
|
|
754
|
-
in_name_prepositions: string[];
|
|
755
|
-
company_suffix_words: string[];
|
|
756
|
-
comma_gated_direct_prefixes: string[];
|
|
757
|
-
};
|
|
758
|
-
type NativeDateData = {
|
|
759
|
-
month_names_by_language: DateMonthData;
|
|
760
|
-
year_words_by_language: YearWordData;
|
|
761
|
-
};
|
|
762
|
-
type NativeMonetaryData = MonetaryData;
|
|
763
|
-
type NativeAddressSeedData = AddressSeedData;
|
|
764
|
-
type NativeAddressContextData = AddressContextData;
|
|
765
|
-
type NativeCoreferencePatternData = {
|
|
766
|
-
pattern: string;
|
|
767
|
-
flags: string;
|
|
203
|
+
flags: string;
|
|
768
204
|
};
|
|
769
205
|
type NativeCoreferenceData = {
|
|
770
206
|
definition_patterns: NativeCoreferencePatternData[];
|
|
@@ -804,6 +240,11 @@ type NativeZoneData = {
|
|
|
804
240
|
section_heading_patterns: NativeZonePatternData[];
|
|
805
241
|
signing_clauses: NativeZoneSigningClauseData[];
|
|
806
242
|
};
|
|
243
|
+
type NativeCountryData = {
|
|
244
|
+
labels: string[];
|
|
245
|
+
isoCodes: string[];
|
|
246
|
+
variants: Array<"name" | "alias" | "alpha3" | "alpha2">;
|
|
247
|
+
};
|
|
807
248
|
type NativeGazetteerData = {
|
|
808
249
|
labels: string[];
|
|
809
250
|
is_fuzzy: boolean[];
|
|
@@ -855,7 +296,7 @@ type NativePreparedSearchConfig = {
|
|
|
855
296
|
deny_list_data?: NativeDenyListMatchData;
|
|
856
297
|
false_positive_filters?: NativeDenyListFilterData;
|
|
857
298
|
gazetteer_data?: NativeGazetteerData;
|
|
858
|
-
country_data?:
|
|
299
|
+
country_data?: NativeCountryData;
|
|
859
300
|
hotword_data?: NativeHotwordRuleData;
|
|
860
301
|
trigger_data?: NativeTriggerData;
|
|
861
302
|
legal_form_data?: NativeLegalFormData;
|
|
@@ -869,35 +310,403 @@ type NativePreparedSearchConfig = {
|
|
|
869
310
|
date_data?: NativeDateData;
|
|
870
311
|
monetary_data?: NativeMonetaryData;
|
|
871
312
|
};
|
|
872
|
-
|
|
873
|
-
|
|
313
|
+
//#endregion
|
|
314
|
+
//#region src/types.d.ts
|
|
315
|
+
/**
|
|
316
|
+
* Fields shared by every entity span in the source text.
|
|
317
|
+
*/
|
|
318
|
+
type EntityBase = {
|
|
319
|
+
start: number;
|
|
320
|
+
end: number;
|
|
321
|
+
label: string;
|
|
322
|
+
text: string;
|
|
323
|
+
score: number;
|
|
324
|
+
sourceDetail?: "custom-deny-list" | "custom-regex" | "gazetteer-extension";
|
|
325
|
+
};
|
|
326
|
+
/**
|
|
327
|
+
* A PII entity span found by a primary detection layer
|
|
328
|
+
* (regex, NER, legal forms, deny list, ...).
|
|
329
|
+
*/
|
|
330
|
+
type DetectedEntity = EntityBase & {
|
|
331
|
+
source: Exclude<DetectionSource, typeof DETECTION_SOURCES.COREFERENCE>;
|
|
332
|
+
};
|
|
333
|
+
/**
|
|
334
|
+
* An alias mention of a previously detected entity: a
|
|
335
|
+
* defined term ("the Seller") or a propagated bare
|
|
336
|
+
* mention ("Acme" after "Acme Corp.").
|
|
337
|
+
*
|
|
338
|
+
* `corefSourceText` is required by construction, so an
|
|
339
|
+
* alias cannot exist without the link back to its source
|
|
340
|
+
* entity. Placeholder numbering reads it to give the
|
|
341
|
+
* alias the same placeholder as the source. The link
|
|
342
|
+
* travels with the entity instead of living in a
|
|
343
|
+
* side-channel map that a producer could forget to
|
|
344
|
+
* write — or that a later pass could clear.
|
|
345
|
+
*/
|
|
346
|
+
type CorefAliasEntity = EntityBase & {
|
|
347
|
+
source: typeof DETECTION_SOURCES.COREFERENCE; /** Full text of the source entity this alias refers to. */
|
|
348
|
+
corefSourceText: string;
|
|
349
|
+
};
|
|
350
|
+
/**
|
|
351
|
+
* A detected PII entity span in the source text.
|
|
352
|
+
* Every detection layer produces these.
|
|
353
|
+
*/
|
|
354
|
+
type Entity = DetectedEntity | CorefAliasEntity;
|
|
355
|
+
/**
|
|
356
|
+
* Entity after human review. Extends the base Entity
|
|
357
|
+
* with a review decision.
|
|
358
|
+
*/
|
|
359
|
+
type ReviewDecision = "confirmed" | "rejected" | "relabeled";
|
|
360
|
+
type ReviewedEntity = Entity & {
|
|
361
|
+
decision?: ReviewDecision;
|
|
362
|
+
originalLabel?: string;
|
|
363
|
+
};
|
|
364
|
+
/**
|
|
365
|
+
* A single entry in the workspace-scoped gazetteer
|
|
366
|
+
* (deny list). Persisted in IndexedDB.
|
|
367
|
+
*/
|
|
368
|
+
type GazetteerEntry = {
|
|
369
|
+
id: string;
|
|
370
|
+
canonical: string;
|
|
371
|
+
label: string;
|
|
372
|
+
variants: string[];
|
|
373
|
+
workspaceId: string;
|
|
374
|
+
createdAt: number;
|
|
375
|
+
source: "manual" | "confirmed-from-model";
|
|
376
|
+
};
|
|
377
|
+
/** Extraction strategy — closed discriminated union. */
|
|
378
|
+
type TriggerStrategy = {
|
|
379
|
+
type: "to-next-comma";
|
|
380
|
+
/**
|
|
381
|
+
* Optional list of lowercase keywords that terminate
|
|
382
|
+
* the value scan, in addition to commas/newlines. Useful
|
|
383
|
+
* for triggers like court names that may continue past
|
|
384
|
+
* a missing comma into adjacent clause text ("Městským
|
|
385
|
+
* soudem v Praze dne 1. 1. 2020"); listing `"dne"` here
|
|
386
|
+
* stops the scan at the date boundary. Matched on a
|
|
387
|
+
* word-boundary, case-insensitive.
|
|
388
|
+
*/
|
|
389
|
+
stopWords?: string[];
|
|
390
|
+
/**
|
|
391
|
+
* Hard cap on the captured span length, in characters,
|
|
392
|
+
* regardless of where the next comma / stop char sits.
|
|
393
|
+
* Use for triggers that label short formulaic phrases
|
|
394
|
+
* ("State of Delaware") and must not absorb the rest
|
|
395
|
+
* of a long forum-selection clause when the comma is
|
|
396
|
+
* sentences away. Falls back to the default 100-char
|
|
397
|
+
* fallback when omitted.
|
|
398
|
+
*/
|
|
399
|
+
maxLength?: number;
|
|
400
|
+
} | {
|
|
401
|
+
type: "to-end-of-line";
|
|
402
|
+
} | {
|
|
403
|
+
type: "n-words";
|
|
404
|
+
count: number;
|
|
405
|
+
} | {
|
|
406
|
+
type: "company-id-value";
|
|
407
|
+
} | {
|
|
408
|
+
type: "address";
|
|
409
|
+
maxChars?: number;
|
|
410
|
+
} | {
|
|
874
411
|
/**
|
|
875
|
-
*
|
|
876
|
-
*
|
|
412
|
+
* Extract the first regex match in the value text.
|
|
413
|
+
* Useful for shape-bounded values that follow a
|
|
414
|
+
* label on the same line as other fields, where
|
|
415
|
+
* `to-end-of-line` would over-capture. The pattern
|
|
416
|
+
* is anchored to the start of the (already
|
|
417
|
+
* leading-whitespace-stripped) value, so use
|
|
418
|
+
* `(?:.*?)` prefix only when intentional.
|
|
877
419
|
*/
|
|
878
|
-
|
|
420
|
+
type: "match-pattern";
|
|
421
|
+
pattern: string;
|
|
422
|
+
flags?: string;
|
|
879
423
|
};
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
424
|
+
/** Validation rules — closed discriminated union. */
|
|
425
|
+
type TriggerValidation = {
|
|
426
|
+
type: "starts-uppercase";
|
|
427
|
+
} | {
|
|
428
|
+
type: "min-length";
|
|
429
|
+
min: number;
|
|
430
|
+
} | {
|
|
431
|
+
type: "max-length";
|
|
432
|
+
max: number;
|
|
433
|
+
} | {
|
|
434
|
+
type: "no-digits";
|
|
435
|
+
} | {
|
|
436
|
+
type: "has-digits";
|
|
437
|
+
} | {
|
|
438
|
+
type: "matches-pattern";
|
|
439
|
+
pattern: string;
|
|
440
|
+
flags?: string;
|
|
441
|
+
}
|
|
442
|
+
/**
|
|
443
|
+
* Run a named stdnum validator (checksum + length)
|
|
444
|
+
* against the captured value. Keeps the trigger
|
|
445
|
+
* path symmetrical with the formatted-regex
|
|
446
|
+
* detectors so e.g. `CPF nº 00000000000` does not
|
|
447
|
+
* survive as a tax-ID entity.
|
|
448
|
+
*/
|
|
449
|
+
| {
|
|
450
|
+
type: "valid-id";
|
|
451
|
+
validator: ValidIdValidator;
|
|
452
|
+
};
|
|
453
|
+
/** Built-in stdnum validators that can be referenced
|
|
454
|
+
* by `valid-id` validations. */
|
|
455
|
+
type ValidIdValidator = "br.cpf" | "br.cnpj" | "us.rtn";
|
|
456
|
+
/** Auto-generated trigger variants — closed set. */
|
|
457
|
+
type TriggerExtension = "add-colon" | "add-trailing-space" | "add-colon-space" | "normalize-spaces";
|
|
458
|
+
/** V2 trigger config entry (JSON shape). */
|
|
459
|
+
type TriggerGroupConfig = {
|
|
460
|
+
id?: string;
|
|
461
|
+
triggers: string[];
|
|
462
|
+
label: string;
|
|
463
|
+
strategy: TriggerStrategy;
|
|
464
|
+
extensions?: TriggerExtension[];
|
|
465
|
+
validations?: TriggerValidation[];
|
|
466
|
+
/** When true, include the trigger text in the
|
|
467
|
+
* entity span (e.g., court names). */
|
|
468
|
+
includeTrigger?: boolean;
|
|
469
|
+
};
|
|
470
|
+
/** Compiled validation with pre-built regex. */
|
|
471
|
+
type CompiledValidation = {
|
|
472
|
+
type: "starts-uppercase";
|
|
473
|
+
re: RegExp;
|
|
474
|
+
} | {
|
|
475
|
+
type: "min-length";
|
|
476
|
+
min: number;
|
|
477
|
+
} | {
|
|
478
|
+
type: "max-length";
|
|
479
|
+
max: number;
|
|
480
|
+
} | {
|
|
481
|
+
type: "no-digits";
|
|
482
|
+
re: RegExp;
|
|
483
|
+
} | {
|
|
484
|
+
type: "has-digits";
|
|
485
|
+
re: RegExp;
|
|
486
|
+
} | {
|
|
487
|
+
type: "matches-pattern";
|
|
488
|
+
re: RegExp;
|
|
489
|
+
} | {
|
|
490
|
+
type: "valid-id";
|
|
491
|
+
validator: ValidIdValidator;
|
|
492
|
+
check: (value: string) => boolean;
|
|
493
|
+
};
|
|
494
|
+
/**
|
|
495
|
+
* Runtime rule — one per trigger string after
|
|
496
|
+
* expansion. Fed to the Aho-Corasick automaton.
|
|
497
|
+
*/
|
|
498
|
+
type TriggerRule = {
|
|
499
|
+
trigger: string;
|
|
500
|
+
label: string;
|
|
501
|
+
strategy: TriggerStrategy;
|
|
502
|
+
validations: CompiledValidation[];
|
|
503
|
+
includeTrigger: boolean;
|
|
504
|
+
};
|
|
505
|
+
/** Per-label operator selection. Key is the entity label. */
|
|
506
|
+
type OperatorConfig = {
|
|
507
|
+
/** Operator per label. Missing labels default to "replace". */operators: Record<string, OperatorType>; /** Custom replacement string for the redact operator. */
|
|
508
|
+
redactString: string;
|
|
509
|
+
};
|
|
510
|
+
/** Whether an operator produces a reversible redaction entry. */
|
|
511
|
+
type OperatorReversibility = "reversible" | "irreversible";
|
|
512
|
+
type AnonymisationOperator = {
|
|
513
|
+
type: OperatorType;
|
|
514
|
+
reversibility: OperatorReversibility;
|
|
515
|
+
/**
|
|
516
|
+
* Apply the operator to a single entity occurrence.
|
|
517
|
+
* Returns the replacement string to embed in the document.
|
|
518
|
+
*/
|
|
519
|
+
apply: (text: string, label: string, placeholder: string, redactString: string) => string;
|
|
520
|
+
};
|
|
521
|
+
/**
|
|
522
|
+
* Redacted document output with stable entity mapping.
|
|
523
|
+
*/
|
|
524
|
+
type RedactionResult = {
|
|
525
|
+
redactedText: string;
|
|
526
|
+
/**
|
|
527
|
+
* Maps placeholder to original text. Only populated for
|
|
528
|
+
* reversible operators (replace). Empty for redact.
|
|
529
|
+
*/
|
|
530
|
+
redactionMap: Map<string, string>; /** Maps placeholder to the operator that produced it. */
|
|
531
|
+
operatorMap: Map<string, OperatorType>;
|
|
532
|
+
entityCount: number;
|
|
533
|
+
};
|
|
534
|
+
/**
|
|
535
|
+
* Configuration for the detection pipeline.
|
|
536
|
+
*/
|
|
537
|
+
type DenyListCategory = "Names" | "Places" | "Addresses" | "Courts" | "Financial" | "Government" | "Healthcare" | "Education" | "Political" | "Organizations" | "International";
|
|
538
|
+
/**
|
|
539
|
+
* Metadata for a single dictionary entry in the
|
|
540
|
+
* deny-list system. Mirrors the shape from
|
|
541
|
+
* the anonymize-data package so consumers can pass
|
|
542
|
+
* pre-loaded data without a runtime dependency.
|
|
543
|
+
*/
|
|
544
|
+
type DictionaryMeta = {
|
|
545
|
+
label: string;
|
|
546
|
+
category: DenyListCategory;
|
|
547
|
+
country: string | null;
|
|
548
|
+
};
|
|
549
|
+
/**
|
|
550
|
+
* Caller-supplied exact terms for deny-list matching.
|
|
551
|
+
* These entries are merged with the published deny-list
|
|
552
|
+
* dictionaries when `enableDenyList` is enabled.
|
|
553
|
+
*/
|
|
554
|
+
type CustomDenyListEntry = {
|
|
555
|
+
value: string;
|
|
556
|
+
label: string;
|
|
557
|
+
variants?: readonly string[];
|
|
558
|
+
};
|
|
559
|
+
/**
|
|
560
|
+
* Caller-supplied regex detector. The pattern is passed
|
|
561
|
+
* to the underlying text-search regex engine, so use its
|
|
562
|
+
* supported regex syntax. Inline flags such as `(?i)` are
|
|
563
|
+
* accepted when supported by that engine.
|
|
564
|
+
*/
|
|
565
|
+
type CustomRegexPattern = {
|
|
566
|
+
pattern: string;
|
|
567
|
+
label: string;
|
|
568
|
+
score?: number;
|
|
569
|
+
preparedArtifactPolicy?: "include" | "omit";
|
|
570
|
+
};
|
|
571
|
+
/**
|
|
572
|
+
* Pre-loaded dictionary data for dependency injection.
|
|
573
|
+
* Consumers that want name/city/deny-list detection
|
|
574
|
+
* load dictionaries themselves (e.g. from the
|
|
575
|
+
* anonymize-data package) and pass them here; the
|
|
576
|
+
* anonymize package has zero cross-package imports.
|
|
577
|
+
*
|
|
578
|
+
* All fields are optional. When a field is absent,
|
|
579
|
+
* the corresponding detection path is skipped (same
|
|
580
|
+
* behavior as when no dictionaries are available).
|
|
581
|
+
*/
|
|
582
|
+
type Dictionaries = {
|
|
583
|
+
/**
|
|
584
|
+
* First names per language code (e.g., "cs", "de").
|
|
585
|
+
* Merged with legacy config names at init time.
|
|
586
|
+
*/
|
|
587
|
+
firstNames?: Readonly<Record<string, readonly string[]>>;
|
|
588
|
+
/**
|
|
589
|
+
* Surnames per language code.
|
|
590
|
+
* Merged with legacy config names at init time.
|
|
591
|
+
*/
|
|
592
|
+
surnames?: Readonly<Record<string, readonly string[]>>;
|
|
593
|
+
/**
|
|
594
|
+
* Non-Western name tokens per locale code
|
|
595
|
+
* (e.g., "in", "ar", "ja-latn", "ko", "zh-latn",
|
|
596
|
+
* "th", "vi", "fil", "id"). Merged with bundled
|
|
597
|
+
* names-nw-*.json data at init time.
|
|
598
|
+
*/
|
|
599
|
+
nonWesternNames?: Readonly<Record<string, readonly string[]>>;
|
|
600
|
+
/**
|
|
601
|
+
* Pre-loaded deny-list dictionaries keyed by
|
|
602
|
+
* dictionary ID (e.g., "courts/CZ", "banks/DE").
|
|
603
|
+
* Each value is the array of terms for that
|
|
604
|
+
* dictionary.
|
|
605
|
+
*/
|
|
606
|
+
denyList?: Readonly<Record<string, readonly string[]>>;
|
|
607
|
+
/**
|
|
608
|
+
* Metadata per dictionary ID. Required when
|
|
609
|
+
* `denyList` is provided so the pipeline knows
|
|
610
|
+
* labels, categories, and country filters.
|
|
611
|
+
*/
|
|
612
|
+
denyListMeta?: Readonly<Record<string, DictionaryMeta>>;
|
|
613
|
+
/**
|
|
614
|
+
* Pre-loaded city names, already merged across
|
|
615
|
+
* all desired countries.
|
|
616
|
+
*
|
|
617
|
+
* Prefer `citiesByCountry` when callers also pass
|
|
618
|
+
* `denyListCountries` / `denyListRegions`; merged
|
|
619
|
+
* city arrays cannot be scoped after injection.
|
|
620
|
+
*/
|
|
621
|
+
cities?: readonly string[];
|
|
622
|
+
/**
|
|
623
|
+
* Pre-loaded city names keyed by ISO 3166-1 alpha-2
|
|
624
|
+
* country code. When provided, the deny-list builder
|
|
625
|
+
* applies `denyListCountries` / `denyListRegions`
|
|
626
|
+
* before adding city patterns to the search automaton.
|
|
627
|
+
*/
|
|
628
|
+
citiesByCountry?: Readonly<Record<string, readonly string[]>>;
|
|
629
|
+
};
|
|
630
|
+
type PipelineConfig = {
|
|
631
|
+
threshold: number;
|
|
632
|
+
enableTriggerPhrases: boolean;
|
|
633
|
+
enableRegex: boolean;
|
|
634
|
+
/**
|
|
635
|
+
* Expected content language codes. When present, these
|
|
636
|
+
* derive default dictionary scopes for name corpus and
|
|
637
|
+
* deny-list matching unless the lower-level scope fields
|
|
638
|
+
* below are set explicitly.
|
|
639
|
+
*/
|
|
640
|
+
languages?: string[];
|
|
641
|
+
/**
|
|
642
|
+
* Convenience form for single-language documents. Ignored
|
|
643
|
+
* when `languages` is also provided.
|
|
644
|
+
*/
|
|
645
|
+
language?: string;
|
|
646
|
+
/**
|
|
647
|
+
* Enables legal-form organization detection.
|
|
648
|
+
* Required for typed callers; legacy untyped
|
|
649
|
+
* callers that omit this field are treated as
|
|
650
|
+
* enabled at runtime for backward compatibility.
|
|
651
|
+
*/
|
|
652
|
+
enableLegalForms: boolean;
|
|
653
|
+
/**
|
|
654
|
+
* Enables first-name/surname/title corpus matching.
|
|
655
|
+
* When deny-list mode is enabled, this also controls
|
|
656
|
+
* whether name-corpus entries are injected into the
|
|
657
|
+
* deny-list search automaton.
|
|
658
|
+
*/
|
|
659
|
+
enableNameCorpus: boolean;
|
|
660
|
+
/**
|
|
661
|
+
* Optional language scope for first-name/surname
|
|
662
|
+
* dictionaries, using the keys present in
|
|
663
|
+
* `dictionaries.firstNames` / `dictionaries.surnames`
|
|
664
|
+
* (for example `["en", "de"]`). When omitted, all
|
|
665
|
+
* injected name languages are used for backward
|
|
666
|
+
* compatibility.
|
|
667
|
+
*/
|
|
668
|
+
nameCorpusLanguages?: string[];
|
|
669
|
+
enableDenyList: boolean;
|
|
670
|
+
denyListCountries?: string[];
|
|
671
|
+
denyListRegions?: string[];
|
|
672
|
+
denyListExcludeCategories?: string[];
|
|
673
|
+
/**
|
|
674
|
+
* Caller-owned exact terms to match through the
|
|
675
|
+
* deny-list layer. Requires `enableDenyList: true`.
|
|
676
|
+
*/
|
|
677
|
+
customDenyList?: readonly CustomDenyListEntry[];
|
|
678
|
+
/**
|
|
679
|
+
* Caller-owned regex detectors. Requires
|
|
680
|
+
* `enableRegex: true`.
|
|
681
|
+
*/
|
|
682
|
+
customRegexes?: readonly CustomRegexPattern[];
|
|
683
|
+
enableGazetteer: boolean;
|
|
684
|
+
/**
|
|
685
|
+
* Detect country names (ISO 3166-1 names, curated
|
|
686
|
+
* aliases, alpha-3 codes). Defaults to true. Names
|
|
687
|
+
* span all manifest languages plus widely-used
|
|
688
|
+
* additions (Dutch, Russian, Chinese, Arabic, etc.).
|
|
689
|
+
*/
|
|
690
|
+
enableCountries?: boolean;
|
|
691
|
+
enableNer: boolean;
|
|
692
|
+
enableConfidenceBoost: boolean;
|
|
693
|
+
enableCoreference: boolean;
|
|
694
|
+
enableZoneClassification?: boolean;
|
|
695
|
+
enableHotwordRules?: boolean;
|
|
696
|
+
/**
|
|
697
|
+
* Requested output labels. An empty array means
|
|
698
|
+
* "do not filter by label" for deterministic
|
|
699
|
+
* detectors; NER falls back to DEFAULT_ENTITY_LABELS.
|
|
700
|
+
*/
|
|
701
|
+
labels: string[];
|
|
702
|
+
workspaceId: string;
|
|
703
|
+
/**
|
|
704
|
+
* Pre-loaded dictionary data for name, deny-list,
|
|
705
|
+
* and city detection. When omitted, dictionary-based
|
|
706
|
+
* detection paths are skipped. Consumers load from
|
|
707
|
+
* the anonymize-data package and pass the data here.
|
|
708
|
+
*/
|
|
709
|
+
dictionaries?: Dictionaries;
|
|
901
710
|
};
|
|
902
711
|
//#endregion
|
|
903
712
|
//#region src/native.d.ts
|
|
@@ -959,6 +768,9 @@ type NativeAnonymizeBinding = {
|
|
|
959
768
|
};
|
|
960
769
|
prepareStaticSearchPackageBytes: (configJson: Uint8Array) => Uint8Array;
|
|
961
770
|
prepareStaticSearchCompressedPackageBytes: (configJson: Uint8Array) => Uint8Array;
|
|
771
|
+
assembleStaticSearchConfigJson?: (pipelineConfigJson: Uint8Array, dictionariesJson?: Uint8Array, gazetteerJson?: Uint8Array) => Uint8Array;
|
|
772
|
+
assembleStaticSearchPackageBytes?: (pipelineConfigJson: Uint8Array, dictionariesJson?: Uint8Array, gazetteerJson?: Uint8Array) => Uint8Array;
|
|
773
|
+
assembleStaticSearchCompressedPackageBytes?: (pipelineConfigJson: Uint8Array, dictionariesJson?: Uint8Array, gazetteerJson?: Uint8Array) => Uint8Array;
|
|
962
774
|
};
|
|
963
775
|
type NativeOperatorConfig = {
|
|
964
776
|
operators?: Record<string, OperatorType>;
|
|
@@ -1154,5 +966,5 @@ type PreparedSearch = PreparedNativeAnonymizer;
|
|
|
1154
966
|
declare const PreparedAnonymizer: typeof PreparedNativeAnonymizer;
|
|
1155
967
|
type PreparedAnonymizer = PreparedNativeAnonymizer;
|
|
1156
968
|
//#endregion
|
|
1157
|
-
export {
|
|
969
|
+
export { OperatorConfig as $, createNativePipelineFromPackage as A, prepare_search_package as B, SharedNativeRedactTextJsonOptions as C, assertNativeBindingVersion as D, SharedNativeSearchPackageOptions as E, getNativeBindingVersion as F, AnonymisationOperator as G, redact_text_json as H, load_prepared_package as I, DenyListCategory as J, CustomDenyListEntry as K, native_package_version as L, diagnostics_stream_json as M, encodeNativeSearchConfig as N, createNativeAnonymizerFromConfig as O, encodeNativeSearchConfigInput as P, GazetteerEntry as Q, normalize_for_search as R, SharedNativePreparedPackageOptions as S, SharedNativeRedactTextStreamJsonOptions as T, redact_text_stream_json as U, redact_text as V, summary_diagnostics_json as W, DictionaryMeta as X, Dictionaries as Y, Entity as Z, PreparedNativeAnonymizer as _, NativeDiagnosticsBatchCallback as a, TriggerGroupConfig as at, SharedNativeDiagnosticsJsonOptions as b, NativePipelineEntity as c, TriggerValidation as ct, NativeRedactionResult as d, PipelineConfig as et, NativeResultEventCallback as f, PreparedAnonymizer as g, NativeStaticRedactionResult as h, NativeBindingVersionOptions as i, TriggerExtension as it, diagnostics_json as j, createNativeAnonymizerFromPackage as k, NativePipelineFromPackageOptions as l, NativePreparedSearchConfig as lt, NativeSearchPackageOptions as m, NativeAnonymizerFromConfigOptions as n, ReviewDecision as nt, NativeNormalizeOptions as o, TriggerRule as ot, NativeSearchPackageInput as p, CustomRegexPattern as q, NativeAnonymizerFromPackageOptions as r, ReviewedEntity as rt, NativeOperatorConfig as s, TriggerStrategy as st, NativeAnonymizeBinding as t, RedactionResult as tt, NativePreparedSearchBinding as u, PreparedNativePipeline as v, SharedNativeRedactTextOptions as w, SharedNativeDiagnosticsStreamJsonOptions as x, PreparedSearch as y, prepareNativeSearchPackage as z };
|
|
1158
970
|
//# sourceMappingURL=native.d.mts.map
|