@stll/anonymize 1.5.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (155) hide show
  1. package/ATTRIBUTION.md +70 -0
  2. package/README.md +86 -48
  3. package/dist/index.d.mts +3 -1201
  4. package/dist/index.mjs +3 -16267
  5. package/dist/index.mjs.map +1 -1
  6. package/dist/native-node.d.mts +138 -0
  7. package/dist/native-node.mjs +3 -0
  8. package/dist/native-node2.d.mts +3 -0
  9. package/dist/native-node2.mjs +728 -0
  10. package/dist/native-node2.mjs.map +1 -0
  11. package/dist/native.d.mts +970 -0
  12. package/dist/native.mjs +230 -0
  13. package/dist/native.mjs.map +1 -0
  14. package/dist/native2.d.mts +2 -0
  15. package/index.cjs +3 -0
  16. package/native-pipeline.cs.stlanonpkg +0 -0
  17. package/native-pipeline.de.stlanonpkg +0 -0
  18. package/native-pipeline.en.stlanonpkg +0 -0
  19. package/native-pipeline.stlanonpkg +0 -0
  20. package/package.json +57 -9
  21. package/scripts/build-native-pipeline-package.mjs +225 -0
  22. package/dist/address-boundaries.mjs +0 -195
  23. package/dist/address-boundaries.mjs.map +0 -1
  24. package/dist/address-prepositions.mjs +0 -182
  25. package/dist/address-prepositions.mjs.map +0 -1
  26. package/dist/address-stop-keywords.mjs +0 -137
  27. package/dist/address-stop-keywords.mjs.map +0 -1
  28. package/dist/address-stopwords.mjs +0 -84
  29. package/dist/address-stopwords.mjs.map +0 -1
  30. package/dist/allow-list.mjs +0 -196
  31. package/dist/allow-list.mjs.map +0 -1
  32. package/dist/clause-noun-heads.mjs +0 -75
  33. package/dist/clause-noun-heads.mjs.map +0 -1
  34. package/dist/common-words-en.mjs +0 -9887
  35. package/dist/common-words-en.mjs.map +0 -1
  36. package/dist/coreference.cs.mjs +0 -14
  37. package/dist/coreference.cs.mjs.map +0 -1
  38. package/dist/coreference.de.mjs +0 -14
  39. package/dist/coreference.de.mjs.map +0 -1
  40. package/dist/coreference.en.mjs +0 -14
  41. package/dist/coreference.en.mjs.map +0 -1
  42. package/dist/coreference.es.mjs +0 -22
  43. package/dist/coreference.es.mjs.map +0 -1
  44. package/dist/coreference.fr.mjs +0 -32
  45. package/dist/coreference.fr.mjs.map +0 -1
  46. package/dist/coreference.it.mjs +0 -27
  47. package/dist/coreference.it.mjs.map +0 -1
  48. package/dist/coreference.pl.mjs +0 -27
  49. package/dist/coreference.pl.mjs.map +0 -1
  50. package/dist/coreference.pt-br.mjs +0 -14
  51. package/dist/coreference.pt-br.mjs.map +0 -1
  52. package/dist/coreference.sk.mjs +0 -27
  53. package/dist/coreference.sk.mjs.map +0 -1
  54. package/dist/currencies.mjs +0 -231
  55. package/dist/currencies.mjs.map +0 -1
  56. package/dist/date-months.mjs +0 -618
  57. package/dist/date-months.mjs.map +0 -1
  58. package/dist/document-structure-headings.mjs +0 -90
  59. package/dist/document-structure-headings.mjs.map +0 -1
  60. package/dist/generic-roles.mjs +0 -244
  61. package/dist/generic-roles.mjs.map +0 -1
  62. package/dist/hotword-rules.mjs +0 -149
  63. package/dist/hotword-rules.mjs.map +0 -1
  64. package/dist/legal-form-leading-clauses.mjs +0 -23
  65. package/dist/legal-form-leading-clauses.mjs.map +0 -1
  66. package/dist/legal-forms.mjs +0 -2115
  67. package/dist/legal-forms.mjs.map +0 -1
  68. package/dist/legal-role-heads.cs.mjs +0 -42
  69. package/dist/legal-role-heads.cs.mjs.map +0 -1
  70. package/dist/legal-role-heads.de.mjs +0 -33
  71. package/dist/legal-role-heads.de.mjs.map +0 -1
  72. package/dist/legal-role-heads.en.mjs +0 -37
  73. package/dist/legal-role-heads.en.mjs.map +0 -1
  74. package/dist/legal-role-heads.es.mjs +0 -54
  75. package/dist/legal-role-heads.es.mjs.map +0 -1
  76. package/dist/legal-role-heads.fr.mjs +0 -72
  77. package/dist/legal-role-heads.fr.mjs.map +0 -1
  78. package/dist/legal-role-heads.it.mjs +0 -68
  79. package/dist/legal-role-heads.it.mjs.map +0 -1
  80. package/dist/legal-role-heads.pl.mjs +0 -84
  81. package/dist/legal-role-heads.pl.mjs.map +0 -1
  82. package/dist/legal-role-heads.pt-br.mjs +0 -63
  83. package/dist/legal-role-heads.pt-br.mjs.map +0 -1
  84. package/dist/legal-role-heads.sk.mjs +0 -80
  85. package/dist/legal-role-heads.sk.mjs.map +0 -1
  86. package/dist/manifest.mjs +0 -69
  87. package/dist/manifest.mjs.map +0 -1
  88. package/dist/names-exclusions.mjs +0 -223
  89. package/dist/names-exclusions.mjs.map +0 -1
  90. package/dist/names-first.mjs +0 -418
  91. package/dist/names-first.mjs.map +0 -1
  92. package/dist/names-nw-ar.mjs +0 -202
  93. package/dist/names-nw-ar.mjs.map +0 -1
  94. package/dist/names-nw-excluded-allcaps.mjs +0 -112
  95. package/dist/names-nw-excluded-allcaps.mjs.map +0 -1
  96. package/dist/names-nw-fil.mjs +0 -202
  97. package/dist/names-nw-fil.mjs.map +0 -1
  98. package/dist/names-nw-id.mjs +0 -210
  99. package/dist/names-nw-id.mjs.map +0 -1
  100. package/dist/names-nw-in.mjs +0 -526
  101. package/dist/names-nw-in.mjs.map +0 -1
  102. package/dist/names-nw-ja-latn.mjs +0 -260
  103. package/dist/names-nw-ja-latn.mjs.map +0 -1
  104. package/dist/names-nw-ko.mjs +0 -162
  105. package/dist/names-nw-ko.mjs.map +0 -1
  106. package/dist/names-nw-th.mjs +0 -188
  107. package/dist/names-nw-th.mjs.map +0 -1
  108. package/dist/names-nw-vi.mjs +0 -151
  109. package/dist/names-nw-vi.mjs.map +0 -1
  110. package/dist/names-nw-zh-latn.mjs +0 -197
  111. package/dist/names-nw-zh-latn.mjs.map +0 -1
  112. package/dist/names-surnames.mjs +0 -113
  113. package/dist/names-surnames.mjs.map +0 -1
  114. package/dist/names-title-tokens.mjs +0 -40
  115. package/dist/names-title-tokens.mjs.map +0 -1
  116. package/dist/person-stopwords.mjs +0 -205
  117. package/dist/person-stopwords.mjs.map +0 -1
  118. package/dist/section-headings.mjs +0 -64
  119. package/dist/section-headings.mjs.map +0 -1
  120. package/dist/sentence-verb-indicators.mjs +0 -232
  121. package/dist/sentence-verb-indicators.mjs.map +0 -1
  122. package/dist/signing-clauses.mjs +0 -78
  123. package/dist/signing-clauses.mjs.map +0 -1
  124. package/dist/stopwords.mjs +0 -9915
  125. package/dist/stopwords.mjs.map +0 -1
  126. package/dist/structural-single-cap-prefixes.mjs +0 -99
  127. package/dist/structural-single-cap-prefixes.mjs.map +0 -1
  128. package/dist/triggers.cs.mjs +0 -569
  129. package/dist/triggers.cs.mjs.map +0 -1
  130. package/dist/triggers.de.mjs +0 -139
  131. package/dist/triggers.de.mjs.map +0 -1
  132. package/dist/triggers.en.mjs +0 -119
  133. package/dist/triggers.en.mjs.map +0 -1
  134. package/dist/triggers.es.mjs +0 -96
  135. package/dist/triggers.es.mjs.map +0 -1
  136. package/dist/triggers.fr.mjs +0 -275
  137. package/dist/triggers.fr.mjs.map +0 -1
  138. package/dist/triggers.global.mjs +0 -79
  139. package/dist/triggers.global.mjs.map +0 -1
  140. package/dist/triggers.hu.mjs +0 -41
  141. package/dist/triggers.hu.mjs.map +0 -1
  142. package/dist/triggers.it.mjs +0 -74
  143. package/dist/triggers.it.mjs.map +0 -1
  144. package/dist/triggers.pl.mjs +0 -271
  145. package/dist/triggers.pl.mjs.map +0 -1
  146. package/dist/triggers.pt-br.mjs +0 -193
  147. package/dist/triggers.pt-br.mjs.map +0 -1
  148. package/dist/triggers.ro.mjs +0 -59
  149. package/dist/triggers.ro.mjs.map +0 -1
  150. package/dist/triggers.sk.mjs +0 -555
  151. package/dist/triggers.sk.mjs.map +0 -1
  152. package/dist/triggers.sv.mjs +0 -58
  153. package/dist/triggers.sv.mjs.map +0 -1
  154. package/dist/year-words.mjs +0 -62
  155. package/dist/year-words.mjs.map +0 -1
package/ATTRIBUTION.md ADDED
@@ -0,0 +1,70 @@
1
+ # Attribution
2
+
3
+ This library builds on ideas and patterns from several open-source
4
+ projects and academic research.
5
+
6
+ ## Prior Art
7
+
8
+ ### Microsoft Presidio (Apache 2.0)
9
+
10
+ - Context-word boosting architecture
11
+ - Structured PII pattern design (IBAN, phone, email)
12
+ - Operator concept (replace vs redact)
13
+ - https://github.com/microsoft/presidio
14
+
15
+ ### GLiNER / GLiNER.js (MIT)
16
+
17
+ - Span-level and token-level NER via ONNX
18
+ - The `gliner/` module is an original implementation informed
19
+ by the GLiNER architecture (arXiv:2311.08526)
20
+ - Processor and decoder logic reimplemented from scratch
21
+ - https://github.com/urchade/GLiNER
22
+
23
+ ### NameTag / MorphoDiTa (ÚFAL, Charles University)
24
+
25
+ - Czech NER and morphological analysis research
26
+ - Czech name declension suffix patterns
27
+ - https://ufal.mff.cuni.cz/nametag
28
+ - https://ufal.mff.cuni.cz/morphodita
29
+
30
+ ### Text Anonymization Benchmark (NorskRegnesentral, MIT)
31
+
32
+ - ECHR court case evaluation methodology
33
+ - Entity type taxonomy for legal documents
34
+ - https://github.com/NorskRegnesentral/text-anonymization-benchmark
35
+
36
+ ### Unicode CLDR
37
+
38
+ - Multilingual month name data
39
+ - https://cldr.unicode.org (Unicode License)
40
+
41
+ ## Deny List Data Sources
42
+
43
+ ### FinNLP/humannames (MIT)
44
+
45
+ - ~195,000 person names (global, multilingual)
46
+ - Used in: `dictionaries/names/global.json`
47
+ - https://github.com/FinNLP/humannames
48
+ - License: MIT
49
+
50
+ ### GeoNames (CC BY 4.0)
51
+
52
+ - City and place names from the GeoNames gazetteer
53
+ - Population threshold: ≥1,000 inhabitants
54
+ - Includes native names, ASCII transliterations, and
55
+ alternate names across languages
56
+ - Used in: `dictionaries/cities/*.json`
57
+ - https://www.geonames.org
58
+ - License: Creative Commons Attribution 4.0 International
59
+
60
+ ### Wikidata (CC0 1.0)
61
+
62
+ - Courts, banks, insurance companies, government ministries,
63
+ universities, hospitals, and EU institutions
64
+ - Labels and alternate labels in cs, sk, de, en
65
+ - Used in: `dictionaries/courts/`, `dictionaries/banks/`,
66
+ `dictionaries/insurance/`, `dictionaries/government/`,
67
+ `dictionaries/education/`, `dictionaries/healthcare/`,
68
+ `dictionaries/international/`
69
+ - https://www.wikidata.org
70
+ - License: Creative Commons CC0 1.0 Universal
package/README.md CHANGED
@@ -1,5 +1,5 @@
1
1
  <p align="center">
2
- <img src="../../.github/assets/banner.png" alt="Stella anonymize" width="100%" />
2
+ <img src="../../.github/assets/banner.png" alt="stella anonymize" width="100%" />
3
3
  </p>
4
4
 
5
5
  # @stll/anonymize
@@ -16,50 +16,96 @@ bun add @stll/anonymize
16
16
  bun add @stll/anonymize-data
17
17
  ```
18
18
 
19
- For browser targets, install `@stll/anonymize-wasm` instead. It exposes the same runtime API through WebAssembly and is the supported entrypoint for Vite-based bundles.
19
+ The Node.js package is Rust-native. Browser/WASM support is maintained through
20
+ `@stll/anonymize-wasm`, which wraps the same native core.
20
21
 
21
- ## Usage
22
+ ## Usage: Node.js native SDK
22
23
 
23
24
  ```ts
24
- import { runPipeline } from "@stll/anonymize";
25
+ import {
26
+ availableDefaultNativePipelineLanguages,
27
+ getDefaultNativePipeline,
28
+ } from "@stll/anonymize/native-node";
29
+
30
+ const languages = availableDefaultNativePipelineLanguages();
31
+ const anonymizer = getDefaultNativePipeline(
32
+ languages.includes("en") ? { language: "en" } : {},
33
+ );
34
+ const result = anonymizer.redact_text(text);
35
+
36
+ console.log(result.redaction.redactedText);
37
+ ```
25
38
 
26
- const entities = await runPipeline({
27
- fullText: text,
28
- config: {
29
- labels: [
30
- "person",
31
- "organization",
32
- "address",
33
- "date",
34
- "iban",
35
- "phone number",
36
- ],
37
- threshold: 0.5,
38
- enableRegex: true,
39
- enableTriggerPhrases: true,
40
- enableLegalForms: true,
41
- enableNameCorpus: true,
42
- enableDenyList: false,
43
- enableGazetteer: false,
44
- enableNer: false,
45
- enableConfidenceBoost: true,
46
- enableCoreference: true,
47
- workspaceId: "default",
48
- },
49
- gazetteerEntries: [],
50
- });
39
+ Call `getDefaultNativePipeline()` once during service startup and reuse the returned anonymizer. The package ships with a prepared native package, so the normal request path avoids rebuilding search automata. Use `preloadDefaultNativePipeline()` or `preloadDefaultNativePipelineAsync()` when the first document should not pay lazy regex warm-up.
40
+
41
+ If your deployment knows the document language up front, select a scoped package at startup. The build emits `en`, `cs`, and `de` scoped packages by default, and `STELLA_ANONYMIZE_NATIVE_PACKAGE_LANGUAGES` can replace that list or be set to an empty value to build only the all-language package:
42
+
43
+ ```bash
44
+ STELLA_ANONYMIZE_NATIVE_PACKAGE_LANGUAGES=en,cs,fr bun run build
51
45
  ```
52
46
 
53
- ## Caller-owned deny lists and regexes
47
+ ```ts
48
+ const anonymizer = getDefaultNativePipeline({ language: "en" });
49
+ ```
50
+
51
+ Regional codes use the exact package when present and otherwise fall back to
52
+ the base language package, so `en-US` can use the shipped `en` artifact.
53
+
54
+ For build-time generated packages or caller-owned data, prepare the package before runtime and load the bytes in the process that handles documents.
54
55
 
55
- Use `customDenyList` for exact terms and variants that you control. These are matched by the deny-list layer, so keep `enableDenyList: true`.
56
+ ```bash
57
+ bunx stella-anonymize-build-native-package \
58
+ --config ./anonymize-native-config.mjs \
59
+ --out ./dist/anonymize.stlanonpkg
60
+ ```
56
61
 
57
62
  ```ts
58
- const entities = await runPipeline({
59
- fullText: text,
63
+ import { load_prepared_package_file } from "@stll/anonymize/native-node";
64
+
65
+ const anonymizer = load_prepared_package_file("./dist/anonymize.stlanonpkg");
66
+ anonymizer.warmLazyRegex();
67
+ const warmDiagnosticsJson = anonymizer.warmLazyRegexDiagnosticsJson();
68
+ const result = anonymizer.redact_text(text, { redactString: "***" });
69
+ ```
70
+
71
+ The config module may export a `PipelineConfig` directly or `{ config, gazetteerEntries }`. Include `@stll/anonymize-data` dictionaries there if your runtime config uses the deny-list or name-corpus layers; keep the corresponding layers enabled for caller-owned `customDenyList`, `customRegexes`, and gazetteers. Those inputs are part of the prepared package and should be regenerated when they change.
72
+
73
+ ## Python SDK
74
+
75
+ ```py
76
+ import stella_anonymize as anonymize
77
+
78
+ languages = anonymize.available_default_native_pipeline_languages()
79
+ prepared = anonymize.preload_default_native_pipeline(
80
+ language="en" if "en" in languages else None
81
+ )
82
+ result = prepared.redact_text(text, redact_string="***")
83
+
84
+ print(result.redaction.redacted_text)
85
+ ```
86
+
87
+ The Python SDK uses the same Rust core and prepared-package contract as the Node SDK. Prefer `get_default_native_pipeline()`, `preload_default_native_pipeline()`, `load_prepared_package()`, or `load_prepared_package_file()` for repeated calls; top-level `redact_text()` and `redact_text_json()` prepare from config on each call.
88
+
89
+ ## Caller-Owned Deny Lists and Regexes
90
+
91
+ Use `customDenyList` for exact terms and variants that you control. Use
92
+ `customRegexes` for deterministic patterns that are not built into the package.
93
+ Caller-owned data is part of the prepared package, so build or load a package
94
+ from that config before serving documents.
95
+
96
+ ```ts
97
+ import {
98
+ createNativePipelineFromConfig,
99
+ loadNativeAnonymizeBinding,
100
+ } from "@stll/anonymize/native-node";
101
+
102
+ const binding = loadNativeAnonymizeBinding();
103
+ const pipeline = await createNativePipelineFromConfig({
104
+ binding,
60
105
  config: {
61
106
  ...baseConfig,
62
107
  enableDenyList: true,
108
+ enableRegex: true,
63
109
  customDenyList: [
64
110
  {
65
111
  value: "Project Nebula",
@@ -67,19 +113,6 @@ const entities = await runPipeline({
67
113
  label: "organization",
68
114
  },
69
115
  ],
70
- },
71
- gazetteerEntries: [],
72
- });
73
- ```
74
-
75
- Use `customRegexes` for deterministic patterns that are not built into the package. These are matched by the regex layer, so keep `enableRegex: true`.
76
-
77
- ```ts
78
- const entities = await runPipeline({
79
- fullText: text,
80
- config: {
81
- ...baseConfig,
82
- enableRegex: true,
83
116
  customRegexes: [
84
117
  {
85
118
  pattern: "\\bSTLL-[0-9]{4}\\b",
@@ -90,6 +123,8 @@ const entities = await runPipeline({
90
123
  },
91
124
  gazetteerEntries: [],
92
125
  });
126
+
127
+ const result = pipeline.redactText(text);
93
128
  ```
94
129
 
95
130
  ## Browser setup
@@ -106,10 +141,13 @@ export default {
106
141
 
107
142
  ## Notes
108
143
 
144
+ - Native architecture and extension guidance:
145
+ [`ARCHITECTURE.md`](ARCHITECTURE.md).
109
146
  - `labels: []` disables deterministic label filtering; when NER is enabled it falls back to the default label set.
110
147
  - `enableNameCorpus` also controls whether first names, surnames, and titles are injected into deny-list matching when `enableDenyList` is enabled.
111
- - The optional `@stll/anonymize-data` package carries the published dictionary and trigger data used by the deny-list layer.
112
- - `customDenyList` and `customRegexes` are part of the pipeline config and are included in the internal search cache key.
148
+ - The optional `@stll/anonymize-data` package carries the published dictionary and trigger data used when building prepared packages.
149
+ - `customDenyList` and `customRegexes` are part of the prepared package input and should be regenerated when they change.
150
+ - The old TypeScript pipeline is kept only as temporary internal migration/test scaffolding under `src/legacy.ts`; it is not the product runtime.
113
151
 
114
152
  ## Built on
115
153