disarm 0.15.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +2 -2
- data/ext/disarm/Cargo.toml +2 -2
- data/ext/disarm/src/lib.rs +200 -55
- data/lib/disarm/version.rb +1 -1
- data/lib/disarm.rb +145 -36
- metadata +3 -3
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 8109d11076864c4647b18efd378ad51e561dc3302ec339f1da1f895253139be6
|
|
4
|
+
data.tar.gz: a42fa1ad22324a6e07bf502e92231a36c628367f720b440ccefe0c3ea4fd82e0
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 764e4f27dc2a2774fdb6df0f4b0cb016f122281dbb44ee8558ce50148427c6a7c23b7b6261004a6a3e64ed931dc0da613e8ebd98fa8b36e58d10b24030a99837
|
|
7
|
+
data.tar.gz: 608a56409c3bdc8ab76f8073880ddd1cd66ca60067063775624fae2306ec83d2591a2011c56f2c329b00f01a09f2db9b16e05331b4b072dbd1977ebec1777369
|
data/README.md
CHANGED
|
@@ -19,10 +19,10 @@ gem "disarm"
|
|
|
19
19
|
gem install disarm
|
|
20
20
|
```
|
|
21
21
|
|
|
22
|
-
Requires Ruby >= 3.
|
|
22
|
+
Requires Ruby >= 3.3. Precompiled platform gems ship for Ruby 3.3 through 4.0
|
|
23
23
|
(Linux x86_64/aarch64, macOS x86_64/arm64, Windows). On a supported Ruby with no
|
|
24
24
|
matching platform gem, the source gem installs and compiles locally, which needs a
|
|
25
|
-
Rust toolchain. Below 3.
|
|
25
|
+
Rust toolchain. Below 3.3 the gem does not install at all.
|
|
26
26
|
|
|
27
27
|
## Usage
|
|
28
28
|
|
data/ext/disarm/Cargo.toml
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
name = "disarm"
|
|
14
14
|
version = "0.0.0"
|
|
15
15
|
edition = "2021"
|
|
16
|
-
rust-version = "1.
|
|
16
|
+
rust-version = "1.88"
|
|
17
17
|
license = "MIT"
|
|
18
18
|
description = "Ruby bindings for disarm — Unicode confusable/text-security building blocks"
|
|
19
19
|
publish = false
|
|
@@ -30,7 +30,7 @@ crate-type = ["cdylib"]
|
|
|
30
30
|
# dep — not a path — so the gem is self-contained and the rb-sys-dock cross-gem
|
|
31
31
|
# build (which mounts only this gem dir) can fetch it. Imported as `disarm_core`
|
|
32
32
|
# because the package name `disarm` would otherwise clash with this crate.
|
|
33
|
-
disarm_core = { package = "disarm", version = "0.
|
|
33
|
+
disarm_core = { package = "disarm", version = "0.17", default-features = false }
|
|
34
34
|
# magnus: ergonomic, safe Ruby<->Rust bindings over rb-sys. 0.8 supports Ruby >= 3.0
|
|
35
35
|
# (the gem's required_ruby_version); rb-sys handles the platform glue.
|
|
36
36
|
magnus = "0.8"
|
data/ext/disarm/src/lib.rs
CHANGED
|
@@ -34,7 +34,8 @@
|
|
|
34
34
|
use std::collections::HashSet;
|
|
35
35
|
|
|
36
36
|
use disarm_core::api;
|
|
37
|
-
use magnus::
|
|
37
|
+
use magnus::encoding::EncodingCapable;
|
|
38
|
+
use magnus::{function, kwargs, method, prelude::*, Error, RHash, RString, Ruby};
|
|
38
39
|
|
|
39
40
|
/// Map a `disarm` error onto the closest standard Ruby exception:
|
|
40
41
|
/// `InvalidArgument` → `ArgumentError`, everything else → `RuntimeError`. The
|
|
@@ -147,9 +148,30 @@ fn wtf8_to_utf8(bytes: &[u8]) -> String {
|
|
|
147
148
|
out
|
|
148
149
|
}
|
|
149
150
|
|
|
150
|
-
/// A text argument decoded at the boundary
|
|
151
|
-
/// Used in place of `String` for every text parameter;
|
|
152
|
-
/// existing `&text` call sites reach the core unchanged.
|
|
151
|
+
/// A text argument decoded at the boundary by the encoding the String declares (#472,
|
|
152
|
+
/// and R1 of `formal/bindings`). Used in place of `String` for every text parameter;
|
|
153
|
+
/// `Deref<Target = str>` lets the existing `&text` call sites reach the core unchanged.
|
|
154
|
+
///
|
|
155
|
+
/// The contract, per `String#encoding`:
|
|
156
|
+
///
|
|
157
|
+
/// - **UTF-8**: the bytes as they are, with the WTF-8 scrub above for malformed input: a
|
|
158
|
+
/// well-formed surrogate pair recombines, each lone surrogate or undecodable byte is
|
|
159
|
+
/// one `U+FFFD`.
|
|
160
|
+
/// - **US-ASCII**: the bytes as they are; a byte above `0x7F` is not US-ASCII and is one
|
|
161
|
+
/// `U+FFFD`.
|
|
162
|
+
/// - **ASCII-8BIT (BINARY)**: bytes with no declared encoding, read as UTF-8 under the
|
|
163
|
+
/// same scrub as a UTF-8 String. `File.binread` and socket reads return BINARY for
|
|
164
|
+
/// what is usually UTF-8, and the C ABI reads its bytes the same way (#1020).
|
|
165
|
+
/// - **Any other encoding** (ISO-8859-1, Windows-1251, Shift_JIS, UTF-16LE, ...):
|
|
166
|
+
/// transcoded to UTF-8 by Ruby's own `String#encode`, each invalid or unmappable
|
|
167
|
+
/// sequence becoming one `U+FFFD`. An encoding Ruby cannot convert from at all (a
|
|
168
|
+
/// dummy such as UTF-7) raises `Encoding::ConverterNotFoundError`, which the Ruby
|
|
169
|
+
/// layer reports as `Disarm::InvalidArgument`.
|
|
170
|
+
///
|
|
171
|
+
/// Until R1 every String's bytes were read as UTF-8 whatever it declared, so an
|
|
172
|
+
/// ISO-8859-1 `"caf\xE9"` transliterated to `"caf[?]"` and a Windows-1251 word to a row
|
|
173
|
+
/// of `[?]`, while the plain `String` parameters of the same functions, which magnus
|
|
174
|
+
/// converts, read the same bytes correctly.
|
|
153
175
|
struct Wtf8Text(String);
|
|
154
176
|
|
|
155
177
|
impl std::ops::Deref for Wtf8Text {
|
|
@@ -161,13 +183,43 @@ impl std::ops::Deref for Wtf8Text {
|
|
|
161
183
|
|
|
162
184
|
impl magnus::TryConvert for Wtf8Text {
|
|
163
185
|
fn try_convert(val: magnus::Value) -> Result<Self, Error> {
|
|
164
|
-
let s =
|
|
186
|
+
let s = RString::try_convert(val)?;
|
|
187
|
+
// GVL invariant: argument conversion runs inside a Ruby method callback.
|
|
188
|
+
#[allow(clippy::expect_used)]
|
|
189
|
+
let ruby = Ruby::get().expect("argument conversion runs while holding the Ruby GVL");
|
|
190
|
+
let encoding = s.enc_get();
|
|
191
|
+
let usascii = encoding == ruby.usascii_encindex();
|
|
192
|
+
let s =
|
|
193
|
+
if encoding == ruby.utf8_encindex() || usascii || encoding == ruby.ascii8bit_encindex()
|
|
194
|
+
{
|
|
195
|
+
s
|
|
196
|
+
} else {
|
|
197
|
+
let options = kwargs!(
|
|
198
|
+
&ruby,
|
|
199
|
+
"invalid" => ruby.to_symbol("replace"),
|
|
200
|
+
"undef" => ruby.to_symbol("replace"),
|
|
201
|
+
"replace" => "\u{FFFD}"
|
|
202
|
+
);
|
|
203
|
+
s.funcall::<_, _, RString>("encode", (ruby.utf8_encoding(), options))?
|
|
204
|
+
};
|
|
165
205
|
// SAFETY: the bytes are copied out immediately; no Ruby API runs between
|
|
166
206
|
// `as_slice` and `to_vec`, so the string cannot be moved or collected.
|
|
167
207
|
let bytes = unsafe { s.as_slice().to_vec() };
|
|
168
208
|
let decoded = match std::str::from_utf8(&bytes) {
|
|
169
|
-
Ok(valid) => valid.to_owned(),
|
|
170
|
-
|
|
209
|
+
Ok(valid) if !usascii || valid.is_ascii() => valid.to_owned(),
|
|
210
|
+
// A US-ASCII String carrying a byte above 0x7F: each such byte is invalid in
|
|
211
|
+
// the encoding the String declares, and is one U+FFFD.
|
|
212
|
+
_ if usascii => bytes
|
|
213
|
+
.iter()
|
|
214
|
+
.map(|&b| {
|
|
215
|
+
if b.is_ascii() {
|
|
216
|
+
char::from(b)
|
|
217
|
+
} else {
|
|
218
|
+
'\u{FFFD}'
|
|
219
|
+
}
|
|
220
|
+
})
|
|
221
|
+
.collect(),
|
|
222
|
+
_ => wtf8_to_utf8(&bytes),
|
|
171
223
|
};
|
|
172
224
|
Ok(Wtf8Text(decoded))
|
|
173
225
|
}
|
|
@@ -191,6 +243,16 @@ fn transliterate_opts(
|
|
|
191
243
|
scheme: String,
|
|
192
244
|
lang: Option<String>,
|
|
193
245
|
) -> Result<String, Error> {
|
|
246
|
+
// `try_run` rejects an unknown `lang` (B2) and applies registered replacements.
|
|
247
|
+
transliterate_builder(&scheme, lang)?
|
|
248
|
+
.try_run(&text)
|
|
249
|
+
.map(std::borrow::Cow::into_owned)
|
|
250
|
+
.map_err(|e| map_err(&e))
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/// The core's `Transliterate` builder for a scheme token and an optional `lang`; the
|
|
254
|
+
/// `lang` is checked when the builder runs.
|
|
255
|
+
fn transliterate_builder(scheme: &str, lang: Option<String>) -> Result<api::Transliterate, Error> {
|
|
194
256
|
let mut builder = api::Transliterate::new();
|
|
195
257
|
if scheme != "default" {
|
|
196
258
|
let scheme: api::Scheme = scheme.parse().map_err(|e| map_err(&e))?;
|
|
@@ -199,7 +261,7 @@ fn transliterate_opts(
|
|
|
199
261
|
if let Some(lang) = lang {
|
|
200
262
|
builder = builder.lang(lang);
|
|
201
263
|
}
|
|
202
|
-
Ok(builder
|
|
264
|
+
Ok(builder)
|
|
203
265
|
}
|
|
204
266
|
|
|
205
267
|
// ── Confusables (TR39) ────────────────────────────────────────────────────────
|
|
@@ -215,6 +277,16 @@ fn normalize_confusables(
|
|
|
215
277
|
Ok(api::normalize_confusables_with(&text, target, digit_policy).into_owned())
|
|
216
278
|
}
|
|
217
279
|
|
|
280
|
+
/// The `digit_policy` token every key builder takes (#896): parsed once, at the boundary.
|
|
281
|
+
fn parse_policy(digit_policy: &str) -> Result<api::DigitPolicy, Error> {
|
|
282
|
+
digit_policy.parse().map_err(|e| map_err(&e))
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/// `Disarm._is_canonical?(text, preset)` — already its own canonical form (#730).
|
|
286
|
+
fn is_canonical(text: Wtf8Text, preset: String) -> Result<bool, Error> {
|
|
287
|
+
api::is_canonical(&text, &preset).map_err(|e| map_err(&e))
|
|
288
|
+
}
|
|
289
|
+
|
|
218
290
|
/// `Disarm._confusable?(text, "latin" | "cyrillic" | "arabic" | "hebrew")`.
|
|
219
291
|
fn is_confusable(text: Wtf8Text, target: String) -> Result<bool, Error> {
|
|
220
292
|
let target: api::TargetScript = target.parse().map_err(|e| map_err(&e))?;
|
|
@@ -300,7 +372,7 @@ fn slugify(
|
|
|
300
372
|
decimal: bool,
|
|
301
373
|
hexadecimal: bool,
|
|
302
374
|
safe_chars: String,
|
|
303
|
-
) -> String {
|
|
375
|
+
) -> Result<String, Error> {
|
|
304
376
|
let mut config = api::SlugConfig::default()
|
|
305
377
|
.with_separator(separator)
|
|
306
378
|
.with_lowercase(lowercase)
|
|
@@ -318,10 +390,17 @@ fn slugify(
|
|
|
318
390
|
config.entities = entities;
|
|
319
391
|
config.decimal = decimal;
|
|
320
392
|
config.hexadecimal = hexadecimal;
|
|
321
|
-
|
|
393
|
+
// `try_slugify` rejects an unknown `lang` rather than falling back (B2).
|
|
394
|
+
api::try_slugify(&text, &config).map_err(|e| map_err(&e))
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
/// `Disarm._replace_emoji(text, replacement)` — every emoji replaced, verbatim (#972).
|
|
398
|
+
fn replace_emoji(text: Wtf8Text, replacement: String) -> String {
|
|
399
|
+
api::replace_emoji(&text, &replacement)
|
|
322
400
|
}
|
|
323
401
|
|
|
324
|
-
/// `Disarm._demojize(text, strip_modifiers)`.
|
|
402
|
+
/// `Disarm._demojize(text, strip_modifiers)`. An emoji CLDR cannot name becomes `[?]`,
|
|
403
|
+
/// as in every binding.
|
|
325
404
|
fn demojize(text: Wtf8Text, strip_modifiers: bool) -> String {
|
|
326
405
|
api::demojize(&text, strip_modifiers)
|
|
327
406
|
}
|
|
@@ -340,8 +419,8 @@ fn demojize(text: Wtf8Text, strip_modifiers: bool) -> String {
|
|
|
340
419
|
///
|
|
341
420
|
/// Unlike `canonicalize` it does NOT fold confusables, so non-Latin text keeps its script
|
|
342
421
|
/// — the point of the preset.
|
|
343
|
-
fn canonicalize_strict(text: Wtf8Text) -> Result<String, Error> {
|
|
344
|
-
api::
|
|
422
|
+
fn canonicalize_strict(text: Wtf8Text, digit_policy: String) -> Result<String, Error> {
|
|
423
|
+
api::canonicalize_strict_with(&text, parse_policy(&digit_policy)?)
|
|
345
424
|
.map(std::borrow::Cow::into_owned)
|
|
346
425
|
.map_err(|e| map_err(&e))
|
|
347
426
|
}
|
|
@@ -350,42 +429,77 @@ fn strip_format(text: Wtf8Text) -> String {
|
|
|
350
429
|
api::strip_format(&text).into_owned()
|
|
351
430
|
}
|
|
352
431
|
|
|
353
|
-
fn strip_obfuscation(text: Wtf8Text) -> Result<String, Error> {
|
|
354
|
-
api::
|
|
432
|
+
fn strip_obfuscation(text: Wtf8Text, digit_policy: String) -> Result<String, Error> {
|
|
433
|
+
api::strip_obfuscation_with(&text, parse_policy(&digit_policy)?)
|
|
355
434
|
.map(std::borrow::Cow::into_owned)
|
|
356
435
|
.map_err(|e| map_err(&e))
|
|
357
436
|
}
|
|
358
437
|
|
|
359
|
-
fn canonicalize(text: Wtf8Text) -> Result<String, Error> {
|
|
360
|
-
api::
|
|
438
|
+
fn canonicalize(text: Wtf8Text, digit_policy: String) -> Result<String, Error> {
|
|
439
|
+
api::canonicalize_with(&text, parse_policy(&digit_policy)?)
|
|
361
440
|
.map(std::borrow::Cow::into_owned)
|
|
362
441
|
.map_err(|e| map_err(&e))
|
|
363
442
|
}
|
|
364
443
|
|
|
365
444
|
/// `Disarm._search_key(text, lang)` — case/accent/script-insensitive lookup key.
|
|
366
445
|
/// `lang` is `nil` (no profile) or a code like `"ru"`. Fails on an unknown `lang`.
|
|
367
|
-
fn search_key(text: Wtf8Text, lang: Option<String
|
|
368
|
-
api::
|
|
446
|
+
fn search_key(text: Wtf8Text, lang: Option<String>, digit_policy: String) -> Result<String, Error> {
|
|
447
|
+
api::search_key_with(&text, lang.as_deref(), parse_policy(&digit_policy)?)
|
|
369
448
|
.map(std::borrow::Cow::into_owned)
|
|
370
449
|
.map_err(|e| map_err(&e))
|
|
371
450
|
}
|
|
372
451
|
|
|
373
452
|
/// `Disarm._sort_key(text, lang)` — collation sort key (preserves base accented
|
|
374
453
|
/// characters for correct ordering). Fails on an unknown `lang`.
|
|
375
|
-
fn sort_key(text: Wtf8Text, lang: Option<String
|
|
376
|
-
api::
|
|
454
|
+
fn sort_key(text: Wtf8Text, lang: Option<String>, digit_policy: String) -> Result<String, Error> {
|
|
455
|
+
api::sort_key_with(&text, lang.as_deref(), parse_policy(&digit_policy)?)
|
|
377
456
|
.map(std::borrow::Cow::into_owned)
|
|
378
457
|
.map_err(|e| map_err(&e))
|
|
379
458
|
}
|
|
380
459
|
|
|
381
460
|
/// `Disarm._catalog_key(text, lang, strict_iso9)` — catalog deduplication key.
|
|
382
461
|
/// `strict_iso9` selects the ISO 9:1995 Cyrillic scheme. Fails on an unknown `lang`.
|
|
383
|
-
fn catalog_key(
|
|
384
|
-
|
|
462
|
+
fn catalog_key(
|
|
463
|
+
text: Wtf8Text,
|
|
464
|
+
lang: Option<String>,
|
|
465
|
+
strict_iso9: bool,
|
|
466
|
+
digit_policy: String,
|
|
467
|
+
) -> Result<String, Error> {
|
|
468
|
+
api::catalog_key_with(
|
|
469
|
+
&text,
|
|
470
|
+
lang.as_deref(),
|
|
471
|
+
strict_iso9,
|
|
472
|
+
parse_policy(&digit_policy)?,
|
|
473
|
+
)
|
|
474
|
+
.map(std::borrow::Cow::into_owned)
|
|
475
|
+
.map_err(|e| map_err(&e))
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
/// `Disarm._skeleton_key(text, digit_policy)` — the TR39 identifier skeleton plus the two
|
|
479
|
+
/// prototype classes disarm's table keeps apart (#650). A spoof key: never for display.
|
|
480
|
+
fn skeleton_key(text: Wtf8Text, digit_policy: String) -> Result<String, Error> {
|
|
481
|
+
api::skeleton_key(&text, parse_policy(&digit_policy)?)
|
|
385
482
|
.map(std::borrow::Cow::into_owned)
|
|
386
483
|
.map_err(|e| map_err(&e))
|
|
387
484
|
}
|
|
388
485
|
|
|
486
|
+
/// `Disarm._edit_distance(a, b)` — Levenshtein distance in characters (#894).
|
|
487
|
+
fn edit_distance(a: Wtf8Text, b: Wtf8Text) -> usize {
|
|
488
|
+
api::edit_distance(&a, &b)
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
/// `Disarm._nearest_match(value, candidates, max_distance)` — the closest candidate and
|
|
492
|
+
/// its distance as a `(value, distance)` tuple, or `nil` beyond `max_distance` (#894).
|
|
493
|
+
/// The Ruby layer maps the tuple to a `{ value:, distance: }` hash.
|
|
494
|
+
fn nearest_match(
|
|
495
|
+
value: Wtf8Text,
|
|
496
|
+
candidates: Vec<String>,
|
|
497
|
+
max_distance: usize,
|
|
498
|
+
) -> Option<(String, usize)> {
|
|
499
|
+
api::nearest_match(&value, candidates.iter().map(String::as_str), max_distance)
|
|
500
|
+
.map(|m| (m.value, m.distance))
|
|
501
|
+
}
|
|
502
|
+
|
|
389
503
|
/// `Disarm._suspicious_hostname?(host)` — flags mixed-script / confusable IDN
|
|
390
504
|
/// spoofs. A false result asserts nothing was *found*, not that the host is safe.
|
|
391
505
|
fn suspicious_hostname(host: Wtf8Text) -> bool {
|
|
@@ -583,16 +697,9 @@ fn find_untranslatable(
|
|
|
583
697
|
scheme: String,
|
|
584
698
|
lang: Option<String>,
|
|
585
699
|
) -> Result<Vec<(String, usize)>, Error> {
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
builder = builder.scheme(scheme);
|
|
590
|
-
}
|
|
591
|
-
if let Some(lang) = lang {
|
|
592
|
-
builder = builder.lang(lang);
|
|
593
|
-
}
|
|
594
|
-
Ok(builder
|
|
595
|
-
.find_untranslatable(&text)
|
|
700
|
+
Ok(transliterate_builder(&scheme, lang)?
|
|
701
|
+
.try_find_untranslatable(&text)
|
|
702
|
+
.map_err(|e| map_err(&e))?
|
|
596
703
|
.into_iter()
|
|
597
704
|
.map(|u| (u.ch.to_string(), u.offset))
|
|
598
705
|
.collect())
|
|
@@ -670,6 +777,22 @@ fn lang_info(code: String) -> Result<RHash, Error> {
|
|
|
670
777
|
Ok(hash)
|
|
671
778
|
}
|
|
672
779
|
|
|
780
|
+
/// `Disarm._confusable_coverage(script)` — TR39 sources whose prototype is in `script`
|
|
781
|
+
/// and how many the bundled tables fold (#963), as a Ruby Hash with symbol keys
|
|
782
|
+
/// (`{ script:, sources:, folded: }`). The denominator `unmapped_confusables` does not
|
|
783
|
+
/// have. Fails (ArgumentError → Disarm::InvalidArgument) on an unknown script.
|
|
784
|
+
fn confusable_coverage(script: String) -> Result<RHash, Error> {
|
|
785
|
+
let row = api::confusable_coverage(&script).map_err(|e| map_err(&e))?;
|
|
786
|
+
// GVL invariant: a Ruby method callback always holds the GVL. Justified.
|
|
787
|
+
#[allow(clippy::expect_used)]
|
|
788
|
+
let ruby = Ruby::get().expect("a Ruby method callback always holds the GVL");
|
|
789
|
+
let hash = ruby.hash_new();
|
|
790
|
+
hash.aset(ruby.to_symbol("script"), row.script)?;
|
|
791
|
+
hash.aset(ruby.to_symbol("sources"), row.sources)?;
|
|
792
|
+
hash.aset(ruby.to_symbol("folded"), row.folded)?;
|
|
793
|
+
Ok(hash)
|
|
794
|
+
}
|
|
795
|
+
|
|
673
796
|
/// `Disarm._script_info(name)` — curated metadata for one script as a Ruby Hash
|
|
674
797
|
/// with symbol keys (`{ name:, default_lang:, example:, context_aware: }`);
|
|
675
798
|
/// `default_lang` is `nil` when the core has none. Fails (ArgumentError →
|
|
@@ -823,6 +946,26 @@ fn pipeline_process(rb_self: &Pipeline, text: Wtf8Text) -> Result<String, Error>
|
|
|
823
946
|
rb_self.inner.process(&text).map_err(|e| map_err(&e))
|
|
824
947
|
}
|
|
825
948
|
|
|
949
|
+
/// `Disarm::Pipeline#purpose` — what the named profile is for, or `nil` (#860).
|
|
950
|
+
fn pipeline_purpose(rb_self: &Pipeline) -> Option<String> {
|
|
951
|
+
rb_self.inner.purpose().map(str::to_owned)
|
|
952
|
+
}
|
|
953
|
+
|
|
954
|
+
/// `Disarm::Pipeline#_with_digit_policy(policy)` — a copy of this pipeline whose confusable
|
|
955
|
+
/// passes fold under `policy` (#646). Fails when the profile has no confusables step and the
|
|
956
|
+
/// policy is not the default: a setting that would never run is refused rather than kept.
|
|
957
|
+
/// The Ruby layer wraps it as `#with_digit_policy` so the error arrives as
|
|
958
|
+
/// `Disarm::InvalidArgument`, the way every module-level call does.
|
|
959
|
+
fn pipeline_with_digit_policy(rb_self: &Pipeline, digit_policy: String) -> Result<Pipeline, Error> {
|
|
960
|
+
Ok(Pipeline {
|
|
961
|
+
inner: rb_self
|
|
962
|
+
.inner
|
|
963
|
+
.clone()
|
|
964
|
+
.with_digit_policy(parse_policy(&digit_policy)?)
|
|
965
|
+
.map_err(|e| map_err(&e))?,
|
|
966
|
+
})
|
|
967
|
+
}
|
|
968
|
+
|
|
826
969
|
/// `Disarm._get_pipeline(profile)` — build a reusable `Disarm::Pipeline` for a
|
|
827
970
|
/// named policy profile. Fails (Disarm::InvalidArgument) on an unknown profile.
|
|
828
971
|
fn get_pipeline(profile: String) -> Result<Pipeline, Error> {
|
|
@@ -837,6 +980,12 @@ fn get_pipeline(profile: String) -> Result<Pipeline, Error> {
|
|
|
837
980
|
fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
838
981
|
let module = ruby.define_module("Disarm")?;
|
|
839
982
|
|
|
983
|
+
// The zalgo defaults, read from the core so the Ruby layer never restates them: it
|
|
984
|
+
// kept a literal cap of 2 after #788 raised the core's to 3 (`formal/bindings`, B1).
|
|
985
|
+
// Defined before `lib/disarm.rb`'s method definitions read them as defaults.
|
|
986
|
+
module.const_set("DEFAULT_ZALGO_THRESHOLD", api::DEFAULT_ZALGO_THRESHOLD)?;
|
|
987
|
+
module.const_set("DEFAULT_ZALGO_MAX_MARKS", api::DEFAULT_ZALGO_MAX_MARKS)?;
|
|
988
|
+
|
|
840
989
|
// Raw, `_`-prefixed shims wrapped by the idiomatic Ruby layer (#357).
|
|
841
990
|
module.define_singleton_method("_transliterate", function!(transliterate, 1))?;
|
|
842
991
|
module.define_singleton_method("_transliterate_opts", function!(transliterate_opts, 3))?;
|
|
@@ -847,25 +996,27 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
847
996
|
module.define_singleton_method("_confusable?", function!(is_confusable, 2))?;
|
|
848
997
|
module.define_singleton_method("_slugify", function!(slugify, 13))?;
|
|
849
998
|
module.define_singleton_method("_demojize", function!(demojize, 2))?;
|
|
850
|
-
module.define_singleton_method("
|
|
999
|
+
module.define_singleton_method("_replace_emoji", function!(replace_emoji, 2))?;
|
|
1000
|
+
module.define_singleton_method("_canonicalize_strict", function!(canonicalize_strict, 2))?;
|
|
851
1001
|
module.define_singleton_method("_strip_format", function!(strip_format, 1))?;
|
|
852
|
-
module.define_singleton_method("_strip_obfuscation", function!(strip_obfuscation,
|
|
853
|
-
module.define_singleton_method("_canonicalize", function!(canonicalize,
|
|
1002
|
+
module.define_singleton_method("_strip_obfuscation", function!(strip_obfuscation, 2))?;
|
|
1003
|
+
module.define_singleton_method("_canonicalize", function!(canonicalize, 2))?;
|
|
1004
|
+
module.define_singleton_method("_is_canonical?", function!(is_canonical, 2))?;
|
|
854
1005
|
|
|
855
1006
|
// Key-derivation presets (#404 Group A parity backfill).
|
|
856
|
-
module.define_singleton_method("_search_key", function!(search_key,
|
|
857
|
-
module.define_singleton_method("_sort_key", function!(sort_key,
|
|
858
|
-
module.define_singleton_method("_catalog_key", function!(catalog_key,
|
|
1007
|
+
module.define_singleton_method("_search_key", function!(search_key, 3))?;
|
|
1008
|
+
module.define_singleton_method("_sort_key", function!(sort_key, 3))?;
|
|
1009
|
+
module.define_singleton_method("_catalog_key", function!(catalog_key, 4))?;
|
|
1010
|
+
module.define_singleton_method("_skeleton_key", function!(skeleton_key, 2))?;
|
|
1011
|
+
module.define_singleton_method("_edit_distance", function!(edit_distance, 2))?;
|
|
1012
|
+
module.define_singleton_method("_nearest_match", function!(nearest_match, 3))?;
|
|
859
1013
|
|
|
860
1014
|
// No options / no symbols, but still wrapped by the Ruby layer so a wrong-type
|
|
861
1015
|
// argument surfaces as Disarm::InvalidArgument rather than a raw TypeError —
|
|
862
1016
|
// keeping `rescue Disarm::Error` exhaustive across the whole public surface.
|
|
863
1017
|
module.define_singleton_method("_strip_accents", function!(strip_accents, 1))?;
|
|
864
1018
|
module.define_singleton_method("_fold_case", function!(fold_case, 1))?;
|
|
865
|
-
module.define_singleton_method(
|
|
866
|
-
"_is_case_fold_stable?",
|
|
867
|
-
function!(is_case_fold_stable, 1),
|
|
868
|
-
)?;
|
|
1019
|
+
module.define_singleton_method("_is_case_fold_stable?", function!(is_case_fold_stable, 1))?;
|
|
869
1020
|
module.define_singleton_method("_find_key_collisions", function!(find_key_collisions, 3))?;
|
|
870
1021
|
module.define_singleton_method("_suspicious_hostname?", function!(suspicious_hostname, 1))?;
|
|
871
1022
|
module.define_singleton_method("_analyze_hostname", function!(analyze_hostname, 2))?;
|
|
@@ -904,10 +1055,7 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
904
1055
|
function!(reverse_transliterate, 2),
|
|
905
1056
|
)?;
|
|
906
1057
|
module.define_singleton_method("_find_untranslatable", function!(find_untranslatable, 3))?;
|
|
907
|
-
module.define_singleton_method(
|
|
908
|
-
"_unmapped_confusables",
|
|
909
|
-
function!(unmapped_confusables, 1),
|
|
910
|
-
)?;
|
|
1058
|
+
module.define_singleton_method("_unmapped_confusables", function!(unmapped_confusables, 1))?;
|
|
911
1059
|
module.define_singleton_method(
|
|
912
1060
|
"_find_unmapped_confusables",
|
|
913
1061
|
function!(find_unmapped_confusables, 2),
|
|
@@ -922,15 +1070,10 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
922
1070
|
// Metadata introspection (#404 phase 3 parity backfill).
|
|
923
1071
|
module.define_singleton_method("_lang_info", function!(lang_info, 1))?;
|
|
924
1072
|
module.define_singleton_method("_script_info", function!(script_info, 1))?;
|
|
925
|
-
module.define_singleton_method(
|
|
926
|
-
|
|
927
|
-
function!(confusables_version, 0),
|
|
928
|
-
)?;
|
|
1073
|
+
module.define_singleton_method("_confusable_coverage", function!(confusable_coverage, 1))?;
|
|
1074
|
+
module.define_singleton_method("_confusables_version", function!(confusables_version, 0))?;
|
|
929
1075
|
module.define_singleton_method("_unicode_version", function!(unicode_version, 0))?;
|
|
930
|
-
module.define_singleton_method(
|
|
931
|
-
"_key_schema_version",
|
|
932
|
-
function!(key_schema_version, 0),
|
|
933
|
-
)?;
|
|
1076
|
+
module.define_singleton_method("_key_schema_version", function!(key_schema_version, 0))?;
|
|
934
1077
|
module.define_singleton_method("_list_scripts", function!(list_scripts, 0))?;
|
|
935
1078
|
module.define_singleton_method("_list_context_langs", function!(list_context_langs, 0))?;
|
|
936
1079
|
|
|
@@ -954,6 +1097,8 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
954
1097
|
// instance method on the wrapped handle. Mirrors `Disarm::Lexicon` above.
|
|
955
1098
|
let pipeline = module.define_class("Pipeline", ruby.class_object())?;
|
|
956
1099
|
pipeline.define_method("process", method!(pipeline_process, 1))?;
|
|
1100
|
+
pipeline.define_method("purpose", method!(pipeline_purpose, 0))?;
|
|
1101
|
+
pipeline.define_method("_with_digit_policy", method!(pipeline_with_digit_policy, 1))?;
|
|
957
1102
|
module.define_singleton_method("_get_pipeline", function!(get_pipeline, 1))?;
|
|
958
1103
|
Ok(())
|
|
959
1104
|
}
|
data/lib/disarm/version.rb
CHANGED
data/lib/disarm.rb
CHANGED
|
@@ -24,6 +24,13 @@ end
|
|
|
24
24
|
# keyword arguments with the core's defaults, symbol tokens (:latin, :default, …),
|
|
25
25
|
# a single transliterate(text, scheme:) entrypoint, and a Disarm::Error hierarchy.
|
|
26
26
|
# Each method is still a thin wrapper over the pure-Rust `disarm` core.
|
|
27
|
+
#
|
|
28
|
+
# Text arguments are read by the encoding the String declares. UTF-8 and US-ASCII are
|
|
29
|
+
# read as they are, and ASCII-8BIT (BINARY) is read as UTF-8, the way the C ABI reads
|
|
30
|
+
# its bytes; any other encoding (ISO-8859-1, Windows-1251, UTF-16LE, ...) is transcoded
|
|
31
|
+
# to UTF-8 first. A malformed or unmappable sequence becomes one U+FFFD, and so does a
|
|
32
|
+
# lone surrogate, the contract every binding shares (#469). An encoding Ruby cannot
|
|
33
|
+
# convert from at all raises Disarm::InvalidArgument.
|
|
27
34
|
module Disarm
|
|
28
35
|
# Base class for every error disarm raises, so consumers can `rescue
|
|
29
36
|
# Disarm::Error`. The native shim raises Ruby's built-in ArgumentError /
|
|
@@ -38,7 +45,8 @@ module Disarm
|
|
|
38
45
|
# Transliterate Unicode text to ASCII. `scheme:` selects the standard:
|
|
39
46
|
# :default (the general-purpose scheme), :strict_iso9, or :gost7034. `lang:`
|
|
40
47
|
# applies a language profile on top of the scheme (e.g. "uk" → Київ → "Kyiv",
|
|
41
|
-
# "de" → ü → "ue"); nil means no profile. Both accept a String or Symbol.
|
|
48
|
+
# "de" → ü → "ue"); nil means no profile. Both accept a String or Symbol. Raises
|
|
49
|
+
# Disarm::InvalidArgument on an unknown lang, as every binding does.
|
|
42
50
|
def transliterate(text, scheme: :default, lang: nil)
|
|
43
51
|
scheme = scheme.to_s
|
|
44
52
|
lang = lang&.to_s
|
|
@@ -53,7 +61,8 @@ module Disarm
|
|
|
53
61
|
end
|
|
54
62
|
end
|
|
55
63
|
|
|
56
|
-
# Fold cross-script confusables toward `target:` (:latin
|
|
64
|
+
# Fold cross-script confusables toward `target:` (:latin, :cyrillic, :arabic or
|
|
65
|
+
# :hebrew).
|
|
57
66
|
#
|
|
58
67
|
# `digit_policy:` selects how non-Latin DIGITS fold (#561).
|
|
59
68
|
#
|
|
@@ -61,9 +70,9 @@ module Disarm
|
|
|
61
70
|
# right for prose, where a Devanagari zero really is a zero.
|
|
62
71
|
#
|
|
63
72
|
# `:tr39` uses upstream's targets, which send most of them to a Latin letter
|
|
64
|
-
# (`०` → `o`; three of the
|
|
73
|
+
# (`०` → `o`; three of the 47 rows fold to `.` or to the two characters `rn`
|
|
65
74
|
# instead). That is what an identifier *skeleton* wants, since its only job is to
|
|
66
|
-
# make two confusable identifiers collide. The two differ on
|
|
75
|
+
# make two confusable identifiers collide. The two differ on 47 rows and agree
|
|
67
76
|
# everywhere else. Scoped to `target: :latin` — the override rows are generated from
|
|
68
77
|
# the Latin table and carry TR39's Latin-script targets, so with `target: :cyrillic`
|
|
69
78
|
# it is a no-op.
|
|
@@ -75,15 +84,16 @@ module Disarm
|
|
|
75
84
|
translate_errors { _normalize_confusables(text, target.to_s, digit_policy.to_s) }
|
|
76
85
|
end
|
|
77
86
|
|
|
78
|
-
# Whether `text` contains a character confusable with `target:` (:latin
|
|
79
|
-
# :cyrillic).
|
|
87
|
+
# Whether `text` contains a character confusable with `target:` (:latin,
|
|
88
|
+
# :cyrillic, :arabic or :hebrew).
|
|
80
89
|
def confusable?(text, target: :latin)
|
|
81
90
|
translate_errors { _confusable?(text, target.to_s) }
|
|
82
91
|
end
|
|
83
92
|
|
|
84
93
|
# Generate a URL-safe slug. Mirrors the core's `SlugConfig` defaults; every
|
|
85
94
|
# option past `text` is keyword-only. (`regex_pattern`/`replacements` are not
|
|
86
|
-
# surfaced yet — see ext/disarm/src/lib.rs.)
|
|
95
|
+
# surfaced yet — see ext/disarm/src/lib.rs.) Raises Disarm::InvalidArgument on an
|
|
96
|
+
# unknown `lang:`.
|
|
87
97
|
def slugify(
|
|
88
98
|
text,
|
|
89
99
|
separator: "-",
|
|
@@ -111,16 +121,34 @@ module Disarm
|
|
|
111
121
|
end
|
|
112
122
|
|
|
113
123
|
# Replace emoji with their plain names (e.g. "👍" → "thumbs up").
|
|
114
|
-
# `strip_modifiers:` drops skin-tone / variation modifiers before naming.
|
|
124
|
+
# `strip_modifiers:` drops skin-tone / variation modifiers before naming. An emoji
|
|
125
|
+
# CLDR cannot name (a regional indicator or a Plane 14 tag character standing
|
|
126
|
+
# alone) becomes "[?]", the sentinel #transliterate writes, as in every binding.
|
|
115
127
|
def demojize(text, strip_modifiers: false)
|
|
116
128
|
translate_errors { _demojize(text, strip_modifiers) }
|
|
117
129
|
end
|
|
118
130
|
|
|
131
|
+
# Replace every emoji with +replacement+, verbatim.
|
|
132
|
+
#
|
|
133
|
+
# The counterpart to +demojize+, and a different question of a different table.
|
|
134
|
+
# +demojize+ asks what CLDR calls a character, so its domain is the name table, which
|
|
135
|
+
# is wider than the emoji: <tt>demojize("x\u2122y")</tt> is "x trade mark y". This
|
|
136
|
+
# asks whether the UCD calls it an emoji — Emoji_Presentation=Yes, an Emoji=Yes base
|
|
137
|
+
# carrying U+FE0F (not an Extended_Pictographic one, so a black star with a selector
|
|
138
|
+
# stays), and the ZWJ, modifier, keycap and flag sequences on those. Nothing
|
|
139
|
+
# else moves.
|
|
140
|
+
#
|
|
141
|
+
# +replacement+ is inserted exactly as given: "" closes an intra-word split and " "
|
|
142
|
+
# keeps two words apart, and no rule serves both.
|
|
143
|
+
def replace_emoji(text, replacement = "")
|
|
144
|
+
translate_errors { _replace_emoji(text, replacement) }
|
|
145
|
+
end
|
|
146
|
+
|
|
119
147
|
# Canonicalize, but raise rather than silently normalize a structural difference
|
|
120
148
|
# away — the half of the pair that lets a caller reject input instead of comparing
|
|
121
149
|
# a value the sender never wrote.
|
|
122
|
-
def canonicalize_strict(text)
|
|
123
|
-
translate_errors { _canonicalize_strict(text) }
|
|
150
|
+
def canonicalize_strict(text, digit_policy: :numeric)
|
|
151
|
+
translate_errors { _canonicalize_strict(text, digit_policy.to_s) }
|
|
124
152
|
end
|
|
125
153
|
|
|
126
154
|
# Strip the non-interchange and invisible classes while KEEPING the script.
|
|
@@ -131,13 +159,13 @@ module Disarm
|
|
|
131
159
|
# keeps the VS15/VS16 presentation selectors after a base, which the naive chain
|
|
132
160
|
# deletes, and it collapses TAB/LF to a space, which the primitives leave alone.
|
|
133
161
|
def strip_format(text)
|
|
134
|
-
_strip_format(text)
|
|
162
|
+
translate_errors { _strip_format(text) }
|
|
135
163
|
end
|
|
136
164
|
|
|
137
165
|
# Remove obfuscation (zero-width, bidi, combining-mark abuse) while keeping
|
|
138
166
|
# legible content.
|
|
139
|
-
def strip_obfuscation(text)
|
|
140
|
-
translate_errors { _strip_obfuscation(text) }
|
|
167
|
+
def strip_obfuscation(text, digit_policy: :numeric)
|
|
168
|
+
translate_errors { _strip_obfuscation(text, digit_policy.to_s) }
|
|
141
169
|
end
|
|
142
170
|
|
|
143
171
|
# Canonicalize text for security-sensitive comparison: strip obfuscation,
|
|
@@ -151,8 +179,8 @@ module Disarm
|
|
|
151
179
|
# delimiter can leave here carrying one. #inspect_anomalies reports it as :confusable
|
|
152
180
|
# WHEN the word also carries an ASCII letter, which is the gate that keeps ordinary
|
|
153
181
|
# non-Latin text from firing; a delimiter-only string is not reported.
|
|
154
|
-
def canonicalize(text)
|
|
155
|
-
translate_errors { _canonicalize(text) }
|
|
182
|
+
def canonicalize(text, digit_policy: :numeric)
|
|
183
|
+
translate_errors { _canonicalize(text, digit_policy.to_s) }
|
|
156
184
|
end
|
|
157
185
|
|
|
158
186
|
# @deprecated Renamed to {#canonicalize} in 0.11 (the +_clean+ name
|
|
@@ -165,22 +193,65 @@ module Disarm
|
|
|
165
193
|
# Case/accent/script-insensitive search lookup key. `lang:` applies a
|
|
166
194
|
# language profile for transliteration (e.g. "ru", "uk"); nil means none.
|
|
167
195
|
# Raises Disarm::InvalidArgument on an unknown lang.
|
|
168
|
-
def search_key(text, lang: nil)
|
|
169
|
-
translate_errors { _search_key(text, lang&.to_s) }
|
|
196
|
+
def search_key(text, lang: nil, digit_policy: :numeric)
|
|
197
|
+
translate_errors { _search_key(text, lang&.to_s, digit_policy.to_s) }
|
|
170
198
|
end
|
|
171
199
|
|
|
172
200
|
# Collation sort key (like #search_key, but keeps base accented characters
|
|
173
201
|
# for correct ordering). `lang:` applies a language profile; nil means none.
|
|
174
202
|
# Raises Disarm::InvalidArgument on an unknown lang.
|
|
175
|
-
def sort_key(text, lang: nil)
|
|
176
|
-
translate_errors { _sort_key(text, lang&.to_s) }
|
|
203
|
+
def sort_key(text, lang: nil, digit_policy: :numeric)
|
|
204
|
+
translate_errors { _sort_key(text, lang&.to_s, digit_policy.to_s) }
|
|
177
205
|
end
|
|
178
206
|
|
|
179
207
|
# Library catalog deduplication key (search_key plus confusable folding).
|
|
180
208
|
# `lang:` applies a language profile; `strict_iso9:` selects the ISO 9:1995
|
|
181
209
|
# Cyrillic scheme. Raises Disarm::InvalidArgument on an unknown lang.
|
|
182
|
-
def catalog_key(text, lang: nil, strict_iso9: false)
|
|
183
|
-
translate_errors { _catalog_key(text, lang&.to_s, strict_iso9) }
|
|
210
|
+
def catalog_key(text, lang: nil, strict_iso9: false, digit_policy: :numeric)
|
|
211
|
+
translate_errors { _catalog_key(text, lang&.to_s, strict_iso9, digit_policy.to_s) }
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
# The TR39 identifier skeleton plus the two prototype classes disarm keeps apart
|
|
215
|
+
# (#650). A spoof key: its only job is to make confusable identifiers collide, and its
|
|
216
|
+
# output is never for display. `digit_policy:` is `:numeric` (the letter half only),
|
|
217
|
+
# `:tr39` (adds `1 ≡ l` and `0 ≡ O`) or `:preserve` (a non-Latin numeral keeps its
|
|
218
|
+
# script).
|
|
219
|
+
def skeleton_key(text, digit_policy: :numeric)
|
|
220
|
+
translate_errors { _skeleton_key(text, digit_policy.to_s) }
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
# Levenshtein edit distance between `a` and `b`, in characters (#894). The one class
|
|
224
|
+
# of registry spoofing the confusable tables deliberately do not model — `paypa1`,
|
|
225
|
+
# `adm1n`. Canonicalize both sides first when composed and decomposed spellings
|
|
226
|
+
# should compare equal.
|
|
227
|
+
def edit_distance(left, right)
|
|
228
|
+
translate_errors { _edit_distance(left, right) }
|
|
229
|
+
end
|
|
230
|
+
|
|
231
|
+
# The candidate closest to `value`, as `{ value:, distance: }`, or nil beyond
|
|
232
|
+
# `max_distance:` (#894). Reports; it does not decide. An exact match is reported with
|
|
233
|
+
# distance 0, and ties go to the first candidate at the lowest distance.
|
|
234
|
+
def nearest_match(value, candidates, max_distance: 1)
|
|
235
|
+
hit = translate_errors do
|
|
236
|
+
# A non-collection is a wrong-typed argument, not a NoMethodError (N2).
|
|
237
|
+
raise ::TypeError, "candidates must be an Array or Enumerable" unless candidates.respond_to?(:map)
|
|
238
|
+
|
|
239
|
+
_nearest_match(value, candidates.map(&:to_s), max_distance)
|
|
240
|
+
end
|
|
241
|
+
hit && { value: hit[0], distance: hit[1] }
|
|
242
|
+
end
|
|
243
|
+
|
|
244
|
+
# Whether `text` is already its own canonical form under `preset:` (#730).
|
|
245
|
+
#
|
|
246
|
+
# The verification-path counterpart to the presets: text in, normalized text out is
|
|
247
|
+
# the generation path, and this is the question a caller asks about bytes that arrive
|
|
248
|
+
# already bound. `has_anomalies?` is not this predicate — 142,760 assigned code points are
|
|
249
|
+
# reported clean by the detector and are not their own canonical form.
|
|
250
|
+
#
|
|
251
|
+
# `preset:` is any name in the preset registry or any profile. Raises
|
|
252
|
+
# Disarm::InvalidArgument on an unknown one.
|
|
253
|
+
def canonical?(text, preset: "canonicalize")
|
|
254
|
+
translate_errors { _is_canonical?(text, preset.to_s) }
|
|
184
255
|
end
|
|
185
256
|
|
|
186
257
|
# Strip diacritics ("café" → "cafe").
|
|
@@ -325,15 +396,18 @@ module Disarm
|
|
|
325
396
|
translate_errors { _strip_pua(text) }
|
|
326
397
|
end
|
|
327
398
|
|
|
328
|
-
# Strip "zalgo" combining-mark stacking, keeping at most `max_marks:`
|
|
329
|
-
# combining
|
|
330
|
-
|
|
399
|
+
# Strip "zalgo" combining-mark stacking, keeping at most `max_marks:` marks of each
|
|
400
|
+
# combining class on one base character. The default is the core's,
|
|
401
|
+
# DEFAULT_ZALGO_MAX_MARKS (3), equal to #zalgo?'s threshold (#788), so this never
|
|
402
|
+
# strips from text #zalgo? declines to flag.
|
|
403
|
+
def strip_zalgo(text, max_marks: DEFAULT_ZALGO_MAX_MARKS)
|
|
331
404
|
translate_errors { _strip_zalgo(text, max_marks) }
|
|
332
405
|
end
|
|
333
406
|
|
|
334
407
|
# Whether `text` looks like zalgo: any base character carries more than
|
|
335
|
-
# `threshold:`
|
|
336
|
-
|
|
408
|
+
# `threshold:` marks of one combining class. The default is the core's,
|
|
409
|
+
# DEFAULT_ZALGO_THRESHOLD (3).
|
|
410
|
+
def zalgo?(text, threshold: DEFAULT_ZALGO_THRESHOLD)
|
|
337
411
|
translate_errors { _zalgo?(text, threshold) }
|
|
338
412
|
end
|
|
339
413
|
|
|
@@ -369,13 +443,15 @@ module Disarm
|
|
|
369
443
|
# Turn arbitrary text into a safe filename. `platform:` is :universal
|
|
370
444
|
# (default), :windows, or :posix; `preserve_extension:` keeps the final
|
|
371
445
|
# extension when truncating to `max_length:`. Raises Disarm::InvalidArgument
|
|
372
|
-
# on an unknown platform
|
|
446
|
+
# on an unknown platform, or on a `separator:` that is not printable, non-space
|
|
447
|
+
# ASCII, contains a character illegal on the platform, or contains a path
|
|
448
|
+
# separator ("/" or "\\"). An empty separator is allowed.
|
|
373
449
|
#
|
|
374
450
|
# A safe *filename*, not a safe URL path segment. "%" is legal in a filename, so one
|
|
375
451
|
# the caller typed is kept — sanitize_filename("..%2Fetc") returns "%2Fetc" — and a
|
|
376
452
|
# consumer that percent-decodes the result must validate AFTER decoding. What this
|
|
377
|
-
# will not do is manufacture one: "%"
|
|
378
|
-
#
|
|
453
|
+
# will not do is manufacture one: every "%" in the output is one the input
|
|
454
|
+
# contained, or part of the separator (#721).
|
|
379
455
|
def sanitize_filename(text, separator: "_", max_length: 255, platform: :universal,
|
|
380
456
|
lang: nil, preserve_extension: true)
|
|
381
457
|
translate_errors do
|
|
@@ -422,8 +498,9 @@ module Disarm
|
|
|
422
498
|
end
|
|
423
499
|
end
|
|
424
500
|
|
|
425
|
-
# ML/NLP normalization: NFKC → emoji→text → transliterate →
|
|
426
|
-
# [case fold] → strip control → strip zero-width →
|
|
501
|
+
# ML/NLP normalization: resolve deletions → NFKC → emoji→text → transliterate →
|
|
502
|
+
# strip accents → emoji→text → [case fold] → strip control → strip zero-width →
|
|
503
|
+
# collapse whitespace → NFC.
|
|
427
504
|
#
|
|
428
505
|
# `fold_case:` defaults to true. Pass false in front of a CASED model — folding is
|
|
429
506
|
# destructive, cannot be undone downstream, and an uncased evaluation harness cannot
|
|
@@ -460,7 +537,7 @@ module Disarm
|
|
|
460
537
|
# anomaly detector's `bidi` kind reports nine of the twelve, holding back LRM, RLM
|
|
461
538
|
# and ALM because a lone directional mark is ordinary in right-to-left text.
|
|
462
539
|
def bidi_control?(text)
|
|
463
|
-
_has_bidi_control?(text)
|
|
540
|
+
translate_errors { _has_bidi_control?(text) }
|
|
464
541
|
end
|
|
465
542
|
|
|
466
543
|
# Explain how `lang: "auto"` detection resolves `text`: a hash with
|
|
@@ -486,6 +563,22 @@ module Disarm
|
|
|
486
563
|
translate_errors { _script_info(name.to_s) }
|
|
487
564
|
end
|
|
488
565
|
|
|
566
|
+
# TR39 confusable sources whose prototype is in +script+, and how many of those
|
|
567
|
+
# disarm's bundled tables fold. Returns a Hash with +:script+, +:sources+ and
|
|
568
|
+
# +:folded+ keys.
|
|
569
|
+
#
|
|
570
|
+
# The denominator +unmapped_confusables+ does not have: that measures one bundled
|
|
571
|
+
# table against the whole 6,565-source population, so a script disarm ships no table
|
|
572
|
+
# for reports a number determined by that absence rather than by its coverage.
|
|
573
|
+
#
|
|
574
|
+
# +:folded+ counts sources any bundled table reaches, not sources folded *toward*
|
|
575
|
+
# this script — Greek is 71 of 159, because the Latin table folds Greek letters that
|
|
576
|
+
# look Latin. A script disarm knows that TR39 never uses as a prototype returns 0 of
|
|
577
|
+
# 0. Raises Disarm::InvalidArgument on an unknown script.
|
|
578
|
+
def confusable_coverage(script)
|
|
579
|
+
translate_errors { _confusable_coverage(script.to_s) }
|
|
580
|
+
end
|
|
581
|
+
|
|
489
582
|
# The Unicode `confusables.txt` release the bundled confusable tables were folded
|
|
490
583
|
# from, e.g. "17.0.0". Not a Unicode version for the library as a whole — the
|
|
491
584
|
# case-folding and width tables track different releases (see docs/provenance.md).
|
|
@@ -574,7 +667,9 @@ module Disarm
|
|
|
574
667
|
# pipe.process("Café") # => "cafe"
|
|
575
668
|
# pipe.process("Köln") # reuse the same handle
|
|
576
669
|
#
|
|
577
|
-
# Disarm::Pipeline#process is the Rust-defined instance method on the handle
|
|
670
|
+
# Disarm::Pipeline#process is the Rust-defined instance method on the handle, and
|
|
671
|
+
# Disarm::Pipeline#with_digit_policy(policy) returns a copy whose confusable passes
|
|
672
|
+
# fold under `policy` (#646); a profile with no confusables step refuses one.
|
|
578
673
|
def get_pipeline(profile)
|
|
579
674
|
translate_errors { _get_pipeline(profile.to_s) }
|
|
580
675
|
end
|
|
@@ -603,16 +698,30 @@ module Disarm
|
|
|
603
698
|
# whole surface. The original backtrace is preserved (passed as the third
|
|
604
699
|
# `raise` argument) so the failing native call site stays visible. A bad
|
|
605
700
|
# argument from the native layer can arrive as ArgumentError (an invalid
|
|
606
|
-
# scheme/target), TypeError (a non-String argument),
|
|
607
|
-
# negative max_length)
|
|
701
|
+
# scheme/target), TypeError (a non-String argument), RangeError (e.g. a
|
|
702
|
+
# negative max_length) or EncodingError (a String in an encoding Ruby cannot
|
|
703
|
+
# transcode to UTF-8) — all map to Disarm::InvalidArgument.
|
|
608
704
|
def translate_errors
|
|
609
705
|
yield
|
|
610
706
|
rescue Error
|
|
611
707
|
raise # already in our hierarchy — don't re-wrap
|
|
612
|
-
rescue ::ArgumentError, ::TypeError, ::RangeError => e
|
|
708
|
+
rescue ::ArgumentError, ::TypeError, ::RangeError, ::EncodingError => e
|
|
613
709
|
raise InvalidArgument, e.message, e.backtrace
|
|
614
710
|
rescue ::RuntimeError => e
|
|
615
711
|
raise Error, e.message, e.backtrace
|
|
616
712
|
end
|
|
617
713
|
end
|
|
714
|
+
|
|
715
|
+
# The reusable handle `Disarm.get_pipeline` returns. `#process` is Rust-defined; this
|
|
716
|
+
# reopens the class for the one method that can fail, so its error arrives as
|
|
717
|
+
# `Disarm::InvalidArgument` the way every module-level call's does.
|
|
718
|
+
class Pipeline
|
|
719
|
+
# A copy of this pipeline whose confusable passes fold under `policy` (#646):
|
|
720
|
+
# `:numeric`, `:tr39` or `:preserve`. Raises Disarm::InvalidArgument when the profile
|
|
721
|
+
# has no confusables step and the policy is not the default — a setting that would
|
|
722
|
+
# never run is refused rather than kept.
|
|
723
|
+
def with_digit_policy(policy)
|
|
724
|
+
Disarm.send(:translate_errors) { _with_digit_policy(policy.to_s) }
|
|
725
|
+
end
|
|
726
|
+
end
|
|
618
727
|
end
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: disarm
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.17.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Richard Quinn
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-09-
|
|
11
|
+
date: 2026-09-26 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: rb_sys
|
|
@@ -130,7 +130,7 @@ required_ruby_version: !ruby/object:Gem::Requirement
|
|
|
130
130
|
requirements:
|
|
131
131
|
- - ">="
|
|
132
132
|
- !ruby/object:Gem::Version
|
|
133
|
-
version: 3.
|
|
133
|
+
version: 3.3.0
|
|
134
134
|
required_rubygems_version: !ruby/object:Gem::Requirement
|
|
135
135
|
requirements:
|
|
136
136
|
- - ">="
|