disarm 0.15.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/binding.d.ts +106 -14
- package/binding.js +300 -103
- package/disarm.darwin-arm64.node +0 -0
- package/disarm.darwin-x64.node +0 -0
- package/disarm.linux-arm64-gnu.node +0 -0
- package/disarm.linux-x64-gnu.node +2 -2
- package/disarm.win32-x64-msvc.node +0 -0
- package/index.d.ts +142 -24
- package/index.js +263 -74
- package/package.json +4 -4
package/disarm.darwin-arm64.node
CHANGED
|
Binary file
|
package/disarm.darwin-x64.node
CHANGED
|
Binary file
|
|
Binary file
|
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
[diffend] Oversized file quarantined before diffing.
|
|
2
2
|
name: package/disarm.linux-x64-gnu.node
|
|
3
|
-
size:
|
|
4
|
-
sha256:
|
|
3
|
+
size: 40558392 bytes
|
|
4
|
+
sha256: 9b9fa65e17615962ffc13a7279596b24597ea6f9db079abf32bdcb19d9ae7ba1
|
|
Binary file
|
package/index.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { Lexicon, Pipeline } from './binding';
|
|
2
|
-
import type { Untranslatable, UnmappedConfusable, AutoLangInspection, KeyCollision, LangMeta, ScriptMeta, Finding as NativeFinding, AnomalyReport as NativeAnomalyReport, HostnameAnalysis as NativeHostnameAnalysis } from './binding';
|
|
3
|
-
export type { Untranslatable, UnmappedConfusable, AutoLangInspection, KeyCollision, LangMeta, ScriptMeta, };
|
|
1
|
+
import { Lexicon as NativeLexicon, Pipeline } from './binding';
|
|
2
|
+
import type { Untranslatable, UnmappedConfusable, AutoLangInspection, KeyCollision, LangMeta, ScriptMeta, ConfusableCoverage, Finding as NativeFinding, AnomalyReport as NativeAnomalyReport, HostnameAnalysis as NativeHostnameAnalysis } from './binding';
|
|
3
|
+
export type { Untranslatable, UnmappedConfusable, AutoLangInspection, KeyCollision, LangMeta, ScriptMeta, ConfusableCoverage, };
|
|
4
4
|
/** Findings from {@link analyzeHostname}. `suspicious` is a maximally
|
|
5
5
|
* conservative screen, not a precise verdict (#549). */
|
|
6
6
|
export type HostnameAnalysis = NativeHostnameAnalysis;
|
|
@@ -9,8 +9,13 @@ export type HostnameAnalysis = NativeHostnameAnalysis;
|
|
|
9
9
|
* `inspectAnomalies` rebuild an internal set from the caller's word array on
|
|
10
10
|
* every call; constructing a `Lexicon` once (`new Lexicon([...])`) and passing
|
|
11
11
|
* it instead builds that set a single time and reuses it across calls.
|
|
12
|
+
*
|
|
13
|
+
* A thin subclass of the native handle, so a bad argument to the constructor throws a
|
|
14
|
+
* {@link DisarmError} like every other entry point (`formal/bindings`, N2).
|
|
12
15
|
*/
|
|
13
|
-
export
|
|
16
|
+
export declare class Lexicon extends NativeLexicon {
|
|
17
|
+
constructor(words: string[]);
|
|
18
|
+
}
|
|
14
19
|
/**
|
|
15
20
|
* A reusable, opaque named-policy-profile pipeline handle (#404). Build it once
|
|
16
21
|
* with {@link getPipeline} for a named profile, then apply it to any number of
|
|
@@ -25,7 +30,7 @@ export { Lexicon };
|
|
|
25
30
|
*/
|
|
26
31
|
export { Pipeline };
|
|
27
32
|
/** The anomaly branch that fired for a finding. */
|
|
28
|
-
export type AnomalyKind = 'invisible' | 'bidi' | 'bidi_mixed' | 'zalgo' | 'mixed_script' | 'leet' | 'segmentation' | 'control' | 'compat_fold' | 'confusable' | 'enclosing_mark' | 'mixed_numbers' | 'duplicate_mark';
|
|
33
|
+
export type AnomalyKind = 'invisible' | 'bidi' | 'bidi_mixed' | 'zalgo' | 'mixed_script' | 'leet' | 'segmentation' | 'control' | 'compat_fold' | 'confusable' | 'enclosing_mark' | 'mixed_numbers' | 'duplicate_mark' | 'deletion' | 'smuggled';
|
|
29
34
|
/**
|
|
30
35
|
* One reason a token is anomalous. Re-typed over the generated {@link NativeFinding}
|
|
31
36
|
* so `kind` is the {@link AnomalyKind} string-union rather than a bare `string`.
|
|
@@ -47,14 +52,17 @@ export declare class DisarmInvalidArgument extends DisarmError {
|
|
|
47
52
|
}
|
|
48
53
|
/** Transliteration scheme: the general-purpose default, ISO 9-style ASCII, or GOST R 7.0.34. */
|
|
49
54
|
export type Scheme = 'default' | 'strict_iso9' | 'gost7034';
|
|
50
|
-
/**
|
|
51
|
-
|
|
55
|
+
/**
|
|
56
|
+
* Confusable-folding target script. `'arabic'` and `'hebrew'` fold toward those scripts
|
|
57
|
+
* (#792); the runtime accepted them before this type admitted them.
|
|
58
|
+
*/
|
|
59
|
+
export type TargetScript = 'latin' | 'cyrillic' | 'arabic' | 'hebrew';
|
|
52
60
|
/**
|
|
53
61
|
* How the fold treats non-Latin digits.
|
|
54
62
|
*
|
|
55
63
|
* `'numeric'` (default) sends them to the ASCII digit — `०` becomes `0` — which is right
|
|
56
64
|
* for prose. `'tr39'` uses upstream's targets, which send most of them to a Latin letter
|
|
57
|
-
* (`०` → `o`), and that is what an identifier *skeleton* wants. The two differ on
|
|
65
|
+
* (`०` → `o`), and that is what an identifier *skeleton* wants. The two differ on 47 rows.
|
|
58
66
|
*
|
|
59
67
|
* Three of those rows do not land on a letter: `٠` and `۰` fold to `.`, and `𑣣` folds to
|
|
60
68
|
* the two characters `rn`. A skeleton feeding a label- or path-shaped key has to allow
|
|
@@ -80,19 +88,36 @@ export interface TransliterateOptions {
|
|
|
80
88
|
/** A language profile applied on top of the scheme (e.g. `'uk'`, `'de'`, or `'auto'`). */
|
|
81
89
|
lang?: string;
|
|
82
90
|
}
|
|
83
|
-
/** Romanize Unicode text to ASCII. */
|
|
91
|
+
/** Romanize Unicode text to ASCII. An unknown `lang` throws {@link DisarmInvalidArgument}. */
|
|
84
92
|
export declare function transliterate(text: string, options?: TransliterateOptions): string;
|
|
85
93
|
/** Reverse-transliterate Latin back to a native script (`'el'`, `'ru'`, or `'uk'`). */
|
|
86
94
|
export declare function reverseTransliterate(text: string, options: {
|
|
87
95
|
lang: ReverseLang;
|
|
88
96
|
}): string;
|
|
89
|
-
/**
|
|
97
|
+
/**
|
|
98
|
+
* Every character in `text` with no romanization, as `{ char, offset }` (byte offset), in
|
|
99
|
+
* order. An unknown `lang` throws {@link DisarmInvalidArgument}.
|
|
100
|
+
*/
|
|
90
101
|
export declare function findUntranslatable(text: string, options?: TransliterateOptions): Untranslatable[];
|
|
91
102
|
/** Fold cross-script confusables toward `target` (default `'latin'`). */
|
|
92
103
|
export declare function normalizeConfusables(text: string, options?: {
|
|
93
104
|
target?: TargetScript;
|
|
94
105
|
digitPolicy?: DigitPolicy;
|
|
95
106
|
}): string;
|
|
107
|
+
/**
|
|
108
|
+
* Whether `text` is already its own canonical form under `preset` (#730).
|
|
109
|
+
*
|
|
110
|
+
* The verification-path counterpart to the presets: text in, normalized text out is the
|
|
111
|
+
* generation path, and this is the question a caller asks about bytes that arrive already
|
|
112
|
+
* bound. `hasAnomalies` is not this predicate — 142,760 assigned code points are reported
|
|
113
|
+
* clean by the detector and are not their own canonical form.
|
|
114
|
+
*
|
|
115
|
+
* `preset` is any name in the preset registry or any profile, defaulting to
|
|
116
|
+
* `'canonicalize'`. An unknown one throws {@link DisarmInvalidArgument}.
|
|
117
|
+
*/
|
|
118
|
+
export declare function isCanonical(text: string, options?: {
|
|
119
|
+
preset?: string;
|
|
120
|
+
}): boolean;
|
|
96
121
|
/** Whether `text` contains a character confusable with `target` (default `'latin'`). */
|
|
97
122
|
export declare function isConfusable(text: string, options?: {
|
|
98
123
|
target?: TargetScript;
|
|
@@ -130,7 +155,10 @@ export interface SlugifyOptions {
|
|
|
130
155
|
hexadecimal?: boolean;
|
|
131
156
|
safeChars?: string;
|
|
132
157
|
}
|
|
133
|
-
/**
|
|
158
|
+
/**
|
|
159
|
+
* Generate a URL-safe slug. Mirrors the core's `SlugConfig` defaults. An unknown `lang`
|
|
160
|
+
* throws {@link DisarmInvalidArgument}.
|
|
161
|
+
*/
|
|
134
162
|
export declare function slugify(text: string, options?: SlugifyOptions): string;
|
|
135
163
|
/** Strip diacritics (`"café"` → `"cafe"`). */
|
|
136
164
|
export declare function stripAccents(text: string): string;
|
|
@@ -169,10 +197,29 @@ export type CollisionKey = 'fold_case' | 'search_key' | 'catalog_key' | 'canonic
|
|
|
169
197
|
export declare function findKeyCollisions(values: string[], key: CollisionKey, options?: {
|
|
170
198
|
lang?: string;
|
|
171
199
|
}): KeyCollision[];
|
|
172
|
-
/**
|
|
200
|
+
/**
|
|
201
|
+
* Replace emoji with their plain names. `stripModifiers` drops skin-tone/variation marks.
|
|
202
|
+
* An emoji CLDR cannot name — a regional indicator or a Plane 14 tag character standing
|
|
203
|
+
* alone — becomes `'[?]'`, the sentinel {@link transliterate} writes, and the default in
|
|
204
|
+
* every binding.
|
|
205
|
+
*/
|
|
173
206
|
export declare function demojize(text: string, options?: {
|
|
174
207
|
stripModifiers?: boolean;
|
|
175
208
|
}): string;
|
|
209
|
+
/** Replace every emoji with `replacement`, verbatim (#972).
|
|
210
|
+
*
|
|
211
|
+
* The counterpart to {@link demojize}, and a different question of a different table.
|
|
212
|
+
* `demojize` asks *what does CLDR call this?*, so its domain is the CLDR name table,
|
|
213
|
+
* which is wider than the emoji: `demojize('x™y')` is `'x trade mark y'`. This asks *is
|
|
214
|
+
* this an emoji by the UCD's properties?* — `Emoji_Presentation=Yes`, an `Emoji=Yes`
|
|
215
|
+
* base carrying `U+FE0F` (not an `Extended_Pictographic` one, so `★` + `U+FE0F` stays),
|
|
216
|
+
* and the ZWJ, modifier, keycap and flag sequences on those. Nothing
|
|
217
|
+
* else moves.
|
|
218
|
+
*
|
|
219
|
+
* `replacement` is inserted exactly as given, with no padding and no whitespace collapse:
|
|
220
|
+
* `''` closes an intra-word split (`aa🔥bb` → `aabb`) and `' '` keeps two words apart
|
|
221
|
+
* (`stop🛑now` → `stop now`), and no rule serves both. */
|
|
222
|
+
export declare function replaceEmoji(text: string, replacement?: string): string;
|
|
176
223
|
/** Apply a Unicode normalization `form` (default `'NFC'`). */
|
|
177
224
|
export declare function normalize(text: string, options?: {
|
|
178
225
|
form?: NormalizationForm;
|
|
@@ -206,11 +253,20 @@ export declare function stripVariationSelectors(text: string): string;
|
|
|
206
253
|
export declare function stripNoncharacters(text: string): string;
|
|
207
254
|
/** Strip every Private Use Area code point (#413). */
|
|
208
255
|
export declare function stripPua(text: string): string;
|
|
209
|
-
/**
|
|
256
|
+
/**
|
|
257
|
+
* Cap the marks of each combining class on one base character at `maxMarks`, a
|
|
258
|
+
* non-negative integer. The default is the core's, 3 — equal to {@link isZalgo}'s
|
|
259
|
+
* threshold (#788), so this never strips from text `isZalgo` declines to flag. It is read
|
|
260
|
+
* from the core rather than restated here: this layer had kept the old cap of 2
|
|
261
|
+
* (`formal/bindings`, B1).
|
|
262
|
+
*/
|
|
210
263
|
export declare function stripZalgo(text: string, options?: {
|
|
211
264
|
maxMarks?: number;
|
|
212
265
|
}): string;
|
|
213
|
-
/**
|
|
266
|
+
/**
|
|
267
|
+
* Whether any base character carries more than `threshold` marks of one combining class.
|
|
268
|
+
* The default is the core's, 3.
|
|
269
|
+
*/
|
|
214
270
|
export declare function isZalgo(text: string, options?: {
|
|
215
271
|
threshold?: number;
|
|
216
272
|
}): boolean;
|
|
@@ -219,7 +275,9 @@ export declare function isZalgo(text: string, options?: {
|
|
|
219
275
|
* the half of the pair that lets a caller reject input instead of comparing a value the
|
|
220
276
|
* sender never wrote.
|
|
221
277
|
*/
|
|
222
|
-
export declare function canonicalizeStrict(text: string
|
|
278
|
+
export declare function canonicalizeStrict(text: string, options?: {
|
|
279
|
+
digitPolicy?: DigitPolicy;
|
|
280
|
+
}): string;
|
|
223
281
|
/**
|
|
224
282
|
* Strip the non-interchange and invisible classes while KEEPING the script.
|
|
225
283
|
*
|
|
@@ -231,17 +289,22 @@ export declare function canonicalizeStrict(text: string): string;
|
|
|
231
289
|
*/
|
|
232
290
|
export declare function stripFormat(text: string): string;
|
|
233
291
|
/** Remove obfuscation (zero-width, bidi, combining-mark abuse, homoglyphs) while keeping legible content. */
|
|
234
|
-
export declare function stripObfuscation(text: string
|
|
292
|
+
export declare function stripObfuscation(text: string, options?: {
|
|
293
|
+
digitPolicy?: DigitPolicy;
|
|
294
|
+
}): string;
|
|
235
295
|
/**
|
|
236
|
-
* Canonicalize text for security-sensitive comparison:
|
|
237
|
-
* → strip invisible classes (#413) → strip control → strip
|
|
238
|
-
*
|
|
239
|
-
* (confusables
|
|
296
|
+
* Canonicalize text for security-sensitive comparison: resolve deletions → NFKC →
|
|
297
|
+
* strip bidi/format → strip invisible classes (#413) → strip control → strip
|
|
298
|
+
* zero-width → collapse whitespace → drop repeated marks → cap combining marks
|
|
299
|
+
* (anti-zalgo) → NFC → confusables and NFC to a fixed point → drop repeated marks
|
|
300
|
+
* → cap combining marks again (the fold is iterated with NFC for idempotency).
|
|
240
301
|
*
|
|
241
302
|
* The name describes the mechanism (Unicode canonicalization for matching), not
|
|
242
303
|
* a safety guarantee — this is not an output sanitizer; encode at the sink.
|
|
243
304
|
*/
|
|
244
|
-
export declare function canonicalize(text: string
|
|
305
|
+
export declare function canonicalize(text: string, options?: {
|
|
306
|
+
digitPolicy?: DigitPolicy;
|
|
307
|
+
}): string;
|
|
245
308
|
/**
|
|
246
309
|
* @deprecated Renamed to {@link canonicalize} in 0.11 (the `*Clean` name
|
|
247
310
|
* overpromised safety); removed in 1.0.
|
|
@@ -264,7 +327,13 @@ export interface SanitizeFilenameOptions {
|
|
|
264
327
|
lang?: string;
|
|
265
328
|
preserveExtension?: boolean;
|
|
266
329
|
}
|
|
267
|
-
/**
|
|
330
|
+
/**
|
|
331
|
+
* Turn arbitrary text into a filesystem-safe filename.
|
|
332
|
+
*
|
|
333
|
+
* `separator` must be printable, non-space ASCII with no character illegal on
|
|
334
|
+
* `platform` and no path separator (`/`, `\`); anything else throws
|
|
335
|
+
* {@link DisarmInvalidArgument}. `''` is allowed.
|
|
336
|
+
*/
|
|
268
337
|
export declare function sanitizeFilename(text: string, options?: SanitizeFilenameOptions): string;
|
|
269
338
|
/**
|
|
270
339
|
* Case/accent/script-insensitive search lookup key (like {@link catalogKey}
|
|
@@ -272,6 +341,7 @@ export declare function sanitizeFilename(text: string, options?: SanitizeFilenam
|
|
|
272
341
|
*/
|
|
273
342
|
export declare function searchKey(text: string, options?: {
|
|
274
343
|
lang?: string;
|
|
344
|
+
digitPolicy?: DigitPolicy;
|
|
275
345
|
}): string;
|
|
276
346
|
/**
|
|
277
347
|
* Collation sort key — like {@link searchKey} but preserves base accented
|
|
@@ -279,6 +349,7 @@ export declare function searchKey(text: string, options?: {
|
|
|
279
349
|
*/
|
|
280
350
|
export declare function sortKey(text: string, options?: {
|
|
281
351
|
lang?: string;
|
|
352
|
+
digitPolicy?: DigitPolicy;
|
|
282
353
|
}): string;
|
|
283
354
|
/**
|
|
284
355
|
* Library catalog deduplication key — like {@link searchKey} plus confusable
|
|
@@ -288,7 +359,38 @@ export declare function sortKey(text: string, options?: {
|
|
|
288
359
|
export declare function catalogKey(text: string, options?: {
|
|
289
360
|
lang?: string;
|
|
290
361
|
strictIso9?: boolean;
|
|
362
|
+
digitPolicy?: DigitPolicy;
|
|
291
363
|
}): string;
|
|
364
|
+
/**
|
|
365
|
+
* The TR39 identifier skeleton plus the two prototype classes disarm's table keeps
|
|
366
|
+
* apart (#650). A spoof key: its only job is to make confusable identifiers collide,
|
|
367
|
+
* and its output is never for display. `digitPolicy` `'numeric'` (default) applies the
|
|
368
|
+
* letter half only; `'tr39'` adds `1 ≡ l` and `0 ≡ O`; `'preserve'` keeps a non-Latin
|
|
369
|
+
* numeral in its script.
|
|
370
|
+
*/
|
|
371
|
+
export declare function skeletonKey(text: string, options?: {
|
|
372
|
+
digitPolicy?: DigitPolicy;
|
|
373
|
+
}): string;
|
|
374
|
+
/**
|
|
375
|
+
* Levenshtein edit distance between `a` and `b`, in characters (#894) — the one class of
|
|
376
|
+
* registry spoofing the confusable tables deliberately do not model (`paypa1`, `adm1n`).
|
|
377
|
+
* Canonicalize both sides first when composed and decomposed spellings should compare
|
|
378
|
+
* equal.
|
|
379
|
+
*/
|
|
380
|
+
export declare function editDistance(a: string, b: string): number;
|
|
381
|
+
/** A candidate and how far {@link nearestMatch} found it from the value asked about. */
|
|
382
|
+
export interface NearestMatch {
|
|
383
|
+
value: string;
|
|
384
|
+
distance: number;
|
|
385
|
+
}
|
|
386
|
+
/**
|
|
387
|
+
* The candidate closest to `value`, with its distance, or `null` beyond `maxDistance`
|
|
388
|
+
* (default 1). Reports; it does not decide. An exact match is reported with distance 0,
|
|
389
|
+
* and ties go to the first candidate at the lowest distance (#894).
|
|
390
|
+
*/
|
|
391
|
+
export declare function nearestMatch(value: string, candidates: string[], options?: {
|
|
392
|
+
maxDistance?: number;
|
|
393
|
+
}): NearestMatch | null;
|
|
292
394
|
/** Options for {@link mlNormalize}. */
|
|
293
395
|
export interface MlNormalizeOptions {
|
|
294
396
|
/** Language code selecting the transliteration table; omit for none. */
|
|
@@ -305,8 +407,9 @@ export interface MlNormalizeOptions {
|
|
|
305
407
|
foldCase?: boolean;
|
|
306
408
|
}
|
|
307
409
|
/**
|
|
308
|
-
* ML/NLP normalization: NFKC → emoji→text → transliterate →
|
|
309
|
-
* [case fold] → strip control → strip zero-width →
|
|
410
|
+
* ML/NLP normalization: resolve deletions → NFKC → emoji→text → transliterate →
|
|
411
|
+
* strip accents → emoji→text → [case fold] → strip control → strip zero-width →
|
|
412
|
+
* collapse whitespace → NFC.
|
|
310
413
|
*
|
|
311
414
|
* Note this folds no confusables — it is not a homoglyph defence at any setting. Put
|
|
312
415
|
* {@link normalizeConfusables} in front of it when a model needs both.
|
|
@@ -370,6 +473,21 @@ export declare function langInfo(code: string): LangMeta;
|
|
|
370
473
|
* unknown name throws {@link DisarmInvalidArgument}.
|
|
371
474
|
*/
|
|
372
475
|
export declare function scriptInfo(name: string): ScriptMeta;
|
|
476
|
+
/** TR39 sources whose prototype is in `script`, and how many of those disarm's bundled
|
|
477
|
+
* tables fold (#963).
|
|
478
|
+
*
|
|
479
|
+
* The denominator {@link unmappedConfusables} does not have. That function measures one
|
|
480
|
+
* bundled table against the whole 6,565-source population, which is the right question
|
|
481
|
+
* for a target disarm ships and a misleading one for a script it does not: Greek reports
|
|
482
|
+
* almost the entire population unmapped, and the number means only "there is no Greek
|
|
483
|
+
* table".
|
|
484
|
+
*
|
|
485
|
+
* `folded` counts sources any bundled table reaches, not sources folded *toward* this
|
|
486
|
+
* script — Greek is 71 of 159 because the Latin table folds Greek letters that look
|
|
487
|
+
* Latin. The grouping uses the UCD's script property, so `"Yi"` and 18 other scripts
|
|
488
|
+
* disarm's own enum does not name are addressable here. A script disarm knows that TR39
|
|
489
|
+
* never uses as a prototype returns 0 of 0. */
|
|
490
|
+
export declare function confusableCoverage(script: string): ConfusableCoverage;
|
|
373
491
|
/**
|
|
374
492
|
* The UCD release disarm's normalizer implements. Not a library-wide Unicode version —
|
|
375
493
|
* the bundled tables track different releases. This is the one integrators ask about,
|