disarm 0.16.0 → 0.17.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +2 -2
- data/ext/disarm/Cargo.toml +2 -2
- data/ext/disarm/src/lib.rs +97 -44
- data/lib/disarm/version.rb +1 -1
- data/lib/disarm.rb +51 -25
- metadata +3 -3
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 359f83c7f8f014a863c70caaa5bbea011530f5dc4866013bdd087ce6de77c7b4
|
|
4
|
+
data.tar.gz: 419a067a9dd9fde079844c8e5047b9be0b88cdb9a0f781a38c3f29a5e120ab10
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 9831ccfdd4b6fb49d8f5eb1df058226403fae7afc2ae933118f516fd1203a6e4dcd2c0d1dafd3ab5c63a081aa16013f95203c5df444a5bd8f1b6a5f479bb937a
|
|
7
|
+
data.tar.gz: a410e078c2b8092815673eaa244d1747a3dbff70e9ddcb789143c77f1617816715ea1dbe8f23fd1bf0231768c704da5c3dd7a197ed91c5af2fa3a7e157a6c741
|
data/README.md
CHANGED
|
@@ -19,10 +19,10 @@ gem "disarm"
|
|
|
19
19
|
gem install disarm
|
|
20
20
|
```
|
|
21
21
|
|
|
22
|
-
Requires Ruby >= 3.
|
|
22
|
+
Requires Ruby >= 3.3. Precompiled platform gems ship for Ruby 3.3 through 4.0
|
|
23
23
|
(Linux x86_64/aarch64, macOS x86_64/arm64, Windows). On a supported Ruby with no
|
|
24
24
|
matching platform gem, the source gem installs and compiles locally, which needs a
|
|
25
|
-
Rust toolchain. Below 3.
|
|
25
|
+
Rust toolchain. Below 3.3 the gem does not install at all.
|
|
26
26
|
|
|
27
27
|
## Usage
|
|
28
28
|
|
data/ext/disarm/Cargo.toml
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
name = "disarm"
|
|
14
14
|
version = "0.0.0"
|
|
15
15
|
edition = "2021"
|
|
16
|
-
rust-version = "1.
|
|
16
|
+
rust-version = "1.88"
|
|
17
17
|
license = "MIT"
|
|
18
18
|
description = "Ruby bindings for disarm — Unicode confusable/text-security building blocks"
|
|
19
19
|
publish = false
|
|
@@ -30,7 +30,7 @@ crate-type = ["cdylib"]
|
|
|
30
30
|
# dep — not a path — so the gem is self-contained and the rb-sys-dock cross-gem
|
|
31
31
|
# build (which mounts only this gem dir) can fetch it. Imported as `disarm_core`
|
|
32
32
|
# because the package name `disarm` would otherwise clash with this crate.
|
|
33
|
-
disarm_core = { package = "disarm", version = "0.
|
|
33
|
+
disarm_core = { package = "disarm", version = "0.17", default-features = false }
|
|
34
34
|
# magnus: ergonomic, safe Ruby<->Rust bindings over rb-sys. 0.8 supports Ruby >= 3.0
|
|
35
35
|
# (the gem's required_ruby_version); rb-sys handles the platform glue.
|
|
36
36
|
magnus = "0.8"
|
data/ext/disarm/src/lib.rs
CHANGED
|
@@ -34,7 +34,8 @@
|
|
|
34
34
|
use std::collections::HashSet;
|
|
35
35
|
|
|
36
36
|
use disarm_core::api;
|
|
37
|
-
use magnus::
|
|
37
|
+
use magnus::encoding::EncodingCapable;
|
|
38
|
+
use magnus::{function, kwargs, method, prelude::*, Error, RHash, RString, Ruby};
|
|
38
39
|
|
|
39
40
|
/// Map a `disarm` error onto the closest standard Ruby exception:
|
|
40
41
|
/// `InvalidArgument` → `ArgumentError`, everything else → `RuntimeError`. The
|
|
@@ -147,9 +148,30 @@ fn wtf8_to_utf8(bytes: &[u8]) -> String {
|
|
|
147
148
|
out
|
|
148
149
|
}
|
|
149
150
|
|
|
150
|
-
/// A text argument decoded at the boundary
|
|
151
|
-
/// Used in place of `String` for every text parameter;
|
|
152
|
-
/// existing `&text` call sites reach the core unchanged.
|
|
151
|
+
/// A text argument decoded at the boundary by the encoding the String declares (#472,
|
|
152
|
+
/// and R1 of `formal/bindings`). Used in place of `String` for every text parameter;
|
|
153
|
+
/// `Deref<Target = str>` lets the existing `&text` call sites reach the core unchanged.
|
|
154
|
+
///
|
|
155
|
+
/// The contract, per `String#encoding`:
|
|
156
|
+
///
|
|
157
|
+
/// - **UTF-8**: the bytes as they are, with the WTF-8 scrub above for malformed input: a
|
|
158
|
+
/// well-formed surrogate pair recombines, each lone surrogate or undecodable byte is
|
|
159
|
+
/// one `U+FFFD`.
|
|
160
|
+
/// - **US-ASCII**: the bytes as they are; a byte above `0x7F` is not US-ASCII and is one
|
|
161
|
+
/// `U+FFFD`.
|
|
162
|
+
/// - **ASCII-8BIT (BINARY)**: bytes with no declared encoding, read as UTF-8 under the
|
|
163
|
+
/// same scrub as a UTF-8 String. `File.binread` and socket reads return BINARY for
|
|
164
|
+
/// what is usually UTF-8, and the C ABI reads its bytes the same way (#1020).
|
|
165
|
+
/// - **Any other encoding** (ISO-8859-1, Windows-1251, Shift_JIS, UTF-16LE, ...):
|
|
166
|
+
/// transcoded to UTF-8 by Ruby's own `String#encode`, each invalid or unmappable
|
|
167
|
+
/// sequence becoming one `U+FFFD`. An encoding Ruby cannot convert from at all (a
|
|
168
|
+
/// dummy such as UTF-7) raises `Encoding::ConverterNotFoundError`, which the Ruby
|
|
169
|
+
/// layer reports as `Disarm::InvalidArgument`.
|
|
170
|
+
///
|
|
171
|
+
/// Until R1 every String's bytes were read as UTF-8 whatever it declared, so an
|
|
172
|
+
/// ISO-8859-1 `"caf\xE9"` transliterated to `"caf[?]"` and a Windows-1251 word to a row
|
|
173
|
+
/// of `[?]`, while the plain `String` parameters of the same functions, which magnus
|
|
174
|
+
/// converts, read the same bytes correctly.
|
|
153
175
|
struct Wtf8Text(String);
|
|
154
176
|
|
|
155
177
|
impl std::ops::Deref for Wtf8Text {
|
|
@@ -161,13 +183,43 @@ impl std::ops::Deref for Wtf8Text {
|
|
|
161
183
|
|
|
162
184
|
impl magnus::TryConvert for Wtf8Text {
|
|
163
185
|
fn try_convert(val: magnus::Value) -> Result<Self, Error> {
|
|
164
|
-
let s =
|
|
186
|
+
let s = RString::try_convert(val)?;
|
|
187
|
+
// GVL invariant: argument conversion runs inside a Ruby method callback.
|
|
188
|
+
#[allow(clippy::expect_used)]
|
|
189
|
+
let ruby = Ruby::get().expect("argument conversion runs while holding the Ruby GVL");
|
|
190
|
+
let encoding = s.enc_get();
|
|
191
|
+
let usascii = encoding == ruby.usascii_encindex();
|
|
192
|
+
let s =
|
|
193
|
+
if encoding == ruby.utf8_encindex() || usascii || encoding == ruby.ascii8bit_encindex()
|
|
194
|
+
{
|
|
195
|
+
s
|
|
196
|
+
} else {
|
|
197
|
+
let options = kwargs!(
|
|
198
|
+
&ruby,
|
|
199
|
+
"invalid" => ruby.to_symbol("replace"),
|
|
200
|
+
"undef" => ruby.to_symbol("replace"),
|
|
201
|
+
"replace" => "\u{FFFD}"
|
|
202
|
+
);
|
|
203
|
+
s.funcall::<_, _, RString>("encode", (ruby.utf8_encoding(), options))?
|
|
204
|
+
};
|
|
165
205
|
// SAFETY: the bytes are copied out immediately; no Ruby API runs between
|
|
166
206
|
// `as_slice` and `to_vec`, so the string cannot be moved or collected.
|
|
167
207
|
let bytes = unsafe { s.as_slice().to_vec() };
|
|
168
208
|
let decoded = match std::str::from_utf8(&bytes) {
|
|
169
|
-
Ok(valid) => valid.to_owned(),
|
|
170
|
-
|
|
209
|
+
Ok(valid) if !usascii || valid.is_ascii() => valid.to_owned(),
|
|
210
|
+
// A US-ASCII String carrying a byte above 0x7F: each such byte is invalid in
|
|
211
|
+
// the encoding the String declares, and is one U+FFFD.
|
|
212
|
+
_ if usascii => bytes
|
|
213
|
+
.iter()
|
|
214
|
+
.map(|&b| {
|
|
215
|
+
if b.is_ascii() {
|
|
216
|
+
char::from(b)
|
|
217
|
+
} else {
|
|
218
|
+
'\u{FFFD}'
|
|
219
|
+
}
|
|
220
|
+
})
|
|
221
|
+
.collect(),
|
|
222
|
+
_ => wtf8_to_utf8(&bytes),
|
|
171
223
|
};
|
|
172
224
|
Ok(Wtf8Text(decoded))
|
|
173
225
|
}
|
|
@@ -191,6 +243,16 @@ fn transliterate_opts(
|
|
|
191
243
|
scheme: String,
|
|
192
244
|
lang: Option<String>,
|
|
193
245
|
) -> Result<String, Error> {
|
|
246
|
+
// `try_run` rejects an unknown `lang` (B2) and applies registered replacements.
|
|
247
|
+
transliterate_builder(&scheme, lang)?
|
|
248
|
+
.try_run(&text)
|
|
249
|
+
.map(std::borrow::Cow::into_owned)
|
|
250
|
+
.map_err(|e| map_err(&e))
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/// The core's `Transliterate` builder for a scheme token and an optional `lang`; the
|
|
254
|
+
/// `lang` is checked when the builder runs.
|
|
255
|
+
fn transliterate_builder(scheme: &str, lang: Option<String>) -> Result<api::Transliterate, Error> {
|
|
194
256
|
let mut builder = api::Transliterate::new();
|
|
195
257
|
if scheme != "default" {
|
|
196
258
|
let scheme: api::Scheme = scheme.parse().map_err(|e| map_err(&e))?;
|
|
@@ -199,7 +261,7 @@ fn transliterate_opts(
|
|
|
199
261
|
if let Some(lang) = lang {
|
|
200
262
|
builder = builder.lang(lang);
|
|
201
263
|
}
|
|
202
|
-
Ok(builder
|
|
264
|
+
Ok(builder)
|
|
203
265
|
}
|
|
204
266
|
|
|
205
267
|
// ── Confusables (TR39) ────────────────────────────────────────────────────────
|
|
@@ -310,7 +372,7 @@ fn slugify(
|
|
|
310
372
|
decimal: bool,
|
|
311
373
|
hexadecimal: bool,
|
|
312
374
|
safe_chars: String,
|
|
313
|
-
) -> String {
|
|
375
|
+
) -> Result<String, Error> {
|
|
314
376
|
let mut config = api::SlugConfig::default()
|
|
315
377
|
.with_separator(separator)
|
|
316
378
|
.with_lowercase(lowercase)
|
|
@@ -328,7 +390,8 @@ fn slugify(
|
|
|
328
390
|
config.entities = entities;
|
|
329
391
|
config.decimal = decimal;
|
|
330
392
|
config.hexadecimal = hexadecimal;
|
|
331
|
-
|
|
393
|
+
// `try_slugify` rejects an unknown `lang` rather than falling back (B2).
|
|
394
|
+
api::try_slugify(&text, &config).map_err(|e| map_err(&e))
|
|
332
395
|
}
|
|
333
396
|
|
|
334
397
|
/// `Disarm._replace_emoji(text, replacement)` — every emoji replaced, verbatim (#972).
|
|
@@ -336,7 +399,8 @@ fn replace_emoji(text: Wtf8Text, replacement: String) -> String {
|
|
|
336
399
|
api::replace_emoji(&text, &replacement)
|
|
337
400
|
}
|
|
338
401
|
|
|
339
|
-
/// `Disarm._demojize(text, strip_modifiers)`.
|
|
402
|
+
/// `Disarm._demojize(text, strip_modifiers)`. An emoji CLDR cannot name becomes `[?]`,
|
|
403
|
+
/// as in every binding.
|
|
340
404
|
fn demojize(text: Wtf8Text, strip_modifiers: bool) -> String {
|
|
341
405
|
api::demojize(&text, strip_modifiers)
|
|
342
406
|
}
|
|
@@ -401,9 +465,14 @@ fn catalog_key(
|
|
|
401
465
|
strict_iso9: bool,
|
|
402
466
|
digit_policy: String,
|
|
403
467
|
) -> Result<String, Error> {
|
|
404
|
-
api::catalog_key_with(
|
|
405
|
-
|
|
406
|
-
.
|
|
468
|
+
api::catalog_key_with(
|
|
469
|
+
&text,
|
|
470
|
+
lang.as_deref(),
|
|
471
|
+
strict_iso9,
|
|
472
|
+
parse_policy(&digit_policy)?,
|
|
473
|
+
)
|
|
474
|
+
.map(std::borrow::Cow::into_owned)
|
|
475
|
+
.map_err(|e| map_err(&e))
|
|
407
476
|
}
|
|
408
477
|
|
|
409
478
|
/// `Disarm._skeleton_key(text, digit_policy)` — the TR39 identifier skeleton plus the two
|
|
@@ -628,16 +697,9 @@ fn find_untranslatable(
|
|
|
628
697
|
scheme: String,
|
|
629
698
|
lang: Option<String>,
|
|
630
699
|
) -> Result<Vec<(String, usize)>, Error> {
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
builder = builder.scheme(scheme);
|
|
635
|
-
}
|
|
636
|
-
if let Some(lang) = lang {
|
|
637
|
-
builder = builder.lang(lang);
|
|
638
|
-
}
|
|
639
|
-
Ok(builder
|
|
640
|
-
.find_untranslatable(&text)
|
|
700
|
+
Ok(transliterate_builder(&scheme, lang)?
|
|
701
|
+
.try_find_untranslatable(&text)
|
|
702
|
+
.map_err(|e| map_err(&e))?
|
|
641
703
|
.into_iter()
|
|
642
704
|
.map(|u| (u.ch.to_string(), u.offset))
|
|
643
705
|
.collect())
|
|
@@ -918,6 +980,12 @@ fn get_pipeline(profile: String) -> Result<Pipeline, Error> {
|
|
|
918
980
|
fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
919
981
|
let module = ruby.define_module("Disarm")?;
|
|
920
982
|
|
|
983
|
+
// The zalgo defaults, read from the core so the Ruby layer never restates them: it
|
|
984
|
+
// kept a literal cap of 2 after #788 raised the core's to 3 (`formal/bindings`, B1).
|
|
985
|
+
// Defined before `lib/disarm.rb`'s method definitions read them as defaults.
|
|
986
|
+
module.const_set("DEFAULT_ZALGO_THRESHOLD", api::DEFAULT_ZALGO_THRESHOLD)?;
|
|
987
|
+
module.const_set("DEFAULT_ZALGO_MAX_MARKS", api::DEFAULT_ZALGO_MAX_MARKS)?;
|
|
988
|
+
|
|
921
989
|
// Raw, `_`-prefixed shims wrapped by the idiomatic Ruby layer (#357).
|
|
922
990
|
module.define_singleton_method("_transliterate", function!(transliterate, 1))?;
|
|
923
991
|
module.define_singleton_method("_transliterate_opts", function!(transliterate_opts, 3))?;
|
|
@@ -948,10 +1016,7 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
948
1016
|
// keeping `rescue Disarm::Error` exhaustive across the whole public surface.
|
|
949
1017
|
module.define_singleton_method("_strip_accents", function!(strip_accents, 1))?;
|
|
950
1018
|
module.define_singleton_method("_fold_case", function!(fold_case, 1))?;
|
|
951
|
-
module.define_singleton_method(
|
|
952
|
-
"_is_case_fold_stable?",
|
|
953
|
-
function!(is_case_fold_stable, 1),
|
|
954
|
-
)?;
|
|
1019
|
+
module.define_singleton_method("_is_case_fold_stable?", function!(is_case_fold_stable, 1))?;
|
|
955
1020
|
module.define_singleton_method("_find_key_collisions", function!(find_key_collisions, 3))?;
|
|
956
1021
|
module.define_singleton_method("_suspicious_hostname?", function!(suspicious_hostname, 1))?;
|
|
957
1022
|
module.define_singleton_method("_analyze_hostname", function!(analyze_hostname, 2))?;
|
|
@@ -990,10 +1055,7 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
990
1055
|
function!(reverse_transliterate, 2),
|
|
991
1056
|
)?;
|
|
992
1057
|
module.define_singleton_method("_find_untranslatable", function!(find_untranslatable, 3))?;
|
|
993
|
-
module.define_singleton_method(
|
|
994
|
-
"_unmapped_confusables",
|
|
995
|
-
function!(unmapped_confusables, 1),
|
|
996
|
-
)?;
|
|
1058
|
+
module.define_singleton_method("_unmapped_confusables", function!(unmapped_confusables, 1))?;
|
|
997
1059
|
module.define_singleton_method(
|
|
998
1060
|
"_find_unmapped_confusables",
|
|
999
1061
|
function!(find_unmapped_confusables, 2),
|
|
@@ -1008,19 +1070,10 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
1008
1070
|
// Metadata introspection (#404 phase 3 parity backfill).
|
|
1009
1071
|
module.define_singleton_method("_lang_info", function!(lang_info, 1))?;
|
|
1010
1072
|
module.define_singleton_method("_script_info", function!(script_info, 1))?;
|
|
1011
|
-
module.define_singleton_method(
|
|
1012
|
-
|
|
1013
|
-
function!(confusable_coverage, 1),
|
|
1014
|
-
)?;
|
|
1015
|
-
module.define_singleton_method(
|
|
1016
|
-
"_confusables_version",
|
|
1017
|
-
function!(confusables_version, 0),
|
|
1018
|
-
)?;
|
|
1073
|
+
module.define_singleton_method("_confusable_coverage", function!(confusable_coverage, 1))?;
|
|
1074
|
+
module.define_singleton_method("_confusables_version", function!(confusables_version, 0))?;
|
|
1019
1075
|
module.define_singleton_method("_unicode_version", function!(unicode_version, 0))?;
|
|
1020
|
-
module.define_singleton_method(
|
|
1021
|
-
"_key_schema_version",
|
|
1022
|
-
function!(key_schema_version, 0),
|
|
1023
|
-
)?;
|
|
1076
|
+
module.define_singleton_method("_key_schema_version", function!(key_schema_version, 0))?;
|
|
1024
1077
|
module.define_singleton_method("_list_scripts", function!(list_scripts, 0))?;
|
|
1025
1078
|
module.define_singleton_method("_list_context_langs", function!(list_context_langs, 0))?;
|
|
1026
1079
|
|
data/lib/disarm/version.rb
CHANGED
data/lib/disarm.rb
CHANGED
|
@@ -24,6 +24,13 @@ end
|
|
|
24
24
|
# keyword arguments with the core's defaults, symbol tokens (:latin, :default, …),
|
|
25
25
|
# a single transliterate(text, scheme:) entrypoint, and a Disarm::Error hierarchy.
|
|
26
26
|
# Each method is still a thin wrapper over the pure-Rust `disarm` core.
|
|
27
|
+
#
|
|
28
|
+
# Text arguments are read by the encoding the String declares. UTF-8 and US-ASCII are
|
|
29
|
+
# read as they are, and ASCII-8BIT (BINARY) is read as UTF-8, the way the C ABI reads
|
|
30
|
+
# its bytes; any other encoding (ISO-8859-1, Windows-1251, UTF-16LE, ...) is transcoded
|
|
31
|
+
# to UTF-8 first. A malformed or unmappable sequence becomes one U+FFFD, and so does a
|
|
32
|
+
# lone surrogate, the contract every binding shares (#469). An encoding Ruby cannot
|
|
33
|
+
# convert from at all raises Disarm::InvalidArgument.
|
|
27
34
|
module Disarm
|
|
28
35
|
# Base class for every error disarm raises, so consumers can `rescue
|
|
29
36
|
# Disarm::Error`. The native shim raises Ruby's built-in ArgumentError /
|
|
@@ -38,7 +45,8 @@ module Disarm
|
|
|
38
45
|
# Transliterate Unicode text to ASCII. `scheme:` selects the standard:
|
|
39
46
|
# :default (the general-purpose scheme), :strict_iso9, or :gost7034. `lang:`
|
|
40
47
|
# applies a language profile on top of the scheme (e.g. "uk" → Київ → "Kyiv",
|
|
41
|
-
# "de" → ü → "ue"); nil means no profile. Both accept a String or Symbol.
|
|
48
|
+
# "de" → ü → "ue"); nil means no profile. Both accept a String or Symbol. Raises
|
|
49
|
+
# Disarm::InvalidArgument on an unknown lang, as every binding does.
|
|
42
50
|
def transliterate(text, scheme: :default, lang: nil)
|
|
43
51
|
scheme = scheme.to_s
|
|
44
52
|
lang = lang&.to_s
|
|
@@ -53,7 +61,8 @@ module Disarm
|
|
|
53
61
|
end
|
|
54
62
|
end
|
|
55
63
|
|
|
56
|
-
# Fold cross-script confusables toward `target:` (:latin
|
|
64
|
+
# Fold cross-script confusables toward `target:` (:latin, :cyrillic, :arabic or
|
|
65
|
+
# :hebrew).
|
|
57
66
|
#
|
|
58
67
|
# `digit_policy:` selects how non-Latin DIGITS fold (#561).
|
|
59
68
|
#
|
|
@@ -61,9 +70,9 @@ module Disarm
|
|
|
61
70
|
# right for prose, where a Devanagari zero really is a zero.
|
|
62
71
|
#
|
|
63
72
|
# `:tr39` uses upstream's targets, which send most of them to a Latin letter
|
|
64
|
-
# (`०` → `o`; three of the
|
|
73
|
+
# (`०` → `o`; three of the 47 rows fold to `.` or to the two characters `rn`
|
|
65
74
|
# instead). That is what an identifier *skeleton* wants, since its only job is to
|
|
66
|
-
# make two confusable identifiers collide. The two differ on
|
|
75
|
+
# make two confusable identifiers collide. The two differ on 47 rows and agree
|
|
67
76
|
# everywhere else. Scoped to `target: :latin` — the override rows are generated from
|
|
68
77
|
# the Latin table and carry TR39's Latin-script targets, so with `target: :cyrillic`
|
|
69
78
|
# it is a no-op.
|
|
@@ -75,15 +84,16 @@ module Disarm
|
|
|
75
84
|
translate_errors { _normalize_confusables(text, target.to_s, digit_policy.to_s) }
|
|
76
85
|
end
|
|
77
86
|
|
|
78
|
-
# Whether `text` contains a character confusable with `target:` (:latin
|
|
79
|
-
# :cyrillic).
|
|
87
|
+
# Whether `text` contains a character confusable with `target:` (:latin,
|
|
88
|
+
# :cyrillic, :arabic or :hebrew).
|
|
80
89
|
def confusable?(text, target: :latin)
|
|
81
90
|
translate_errors { _confusable?(text, target.to_s) }
|
|
82
91
|
end
|
|
83
92
|
|
|
84
93
|
# Generate a URL-safe slug. Mirrors the core's `SlugConfig` defaults; every
|
|
85
94
|
# option past `text` is keyword-only. (`regex_pattern`/`replacements` are not
|
|
86
|
-
# surfaced yet — see ext/disarm/src/lib.rs.)
|
|
95
|
+
# surfaced yet — see ext/disarm/src/lib.rs.) Raises Disarm::InvalidArgument on an
|
|
96
|
+
# unknown `lang:`.
|
|
87
97
|
def slugify(
|
|
88
98
|
text,
|
|
89
99
|
separator: "-",
|
|
@@ -111,7 +121,9 @@ module Disarm
|
|
|
111
121
|
end
|
|
112
122
|
|
|
113
123
|
# Replace emoji with their plain names (e.g. "👍" → "thumbs up").
|
|
114
|
-
# `strip_modifiers:` drops skin-tone / variation modifiers before naming.
|
|
124
|
+
# `strip_modifiers:` drops skin-tone / variation modifiers before naming. An emoji
|
|
125
|
+
# CLDR cannot name (a regional indicator or a Plane 14 tag character standing
|
|
126
|
+
# alone) becomes "[?]", the sentinel #transliterate writes, as in every binding.
|
|
115
127
|
def demojize(text, strip_modifiers: false)
|
|
116
128
|
translate_errors { _demojize(text, strip_modifiers) }
|
|
117
129
|
end
|
|
@@ -122,7 +134,8 @@ module Disarm
|
|
|
122
134
|
# +demojize+ asks what CLDR calls a character, so its domain is the name table, which
|
|
123
135
|
# is wider than the emoji: <tt>demojize("x\u2122y")</tt> is "x trade mark y". This
|
|
124
136
|
# asks whether the UCD calls it an emoji — Emoji_Presentation=Yes, an Emoji=Yes base
|
|
125
|
-
# carrying U+FE0F
|
|
137
|
+
# carrying U+FE0F (not an Extended_Pictographic one, so a black star with a selector
|
|
138
|
+
# stays), and the ZWJ, modifier, keycap and flag sequences on those. Nothing
|
|
126
139
|
# else moves.
|
|
127
140
|
#
|
|
128
141
|
# +replacement+ is inserted exactly as given: "" closes an intra-word split and " "
|
|
@@ -146,7 +159,7 @@ module Disarm
|
|
|
146
159
|
# keeps the VS15/VS16 presentation selectors after a base, which the naive chain
|
|
147
160
|
# deletes, and it collapses TAB/LF to a space, which the primitives leave alone.
|
|
148
161
|
def strip_format(text)
|
|
149
|
-
_strip_format(text)
|
|
162
|
+
translate_errors { _strip_format(text) }
|
|
150
163
|
end
|
|
151
164
|
|
|
152
165
|
# Remove obfuscation (zero-width, bidi, combining-mark abuse) while keeping
|
|
@@ -219,7 +232,12 @@ module Disarm
|
|
|
219
232
|
# `max_distance:` (#894). Reports; it does not decide. An exact match is reported with
|
|
220
233
|
# distance 0, and ties go to the first candidate at the lowest distance.
|
|
221
234
|
def nearest_match(value, candidates, max_distance: 1)
|
|
222
|
-
hit = translate_errors
|
|
235
|
+
hit = translate_errors do
|
|
236
|
+
# A non-collection is a wrong-typed argument, not a NoMethodError (N2).
|
|
237
|
+
raise ::TypeError, "candidates must be an Array or Enumerable" unless candidates.respond_to?(:map)
|
|
238
|
+
|
|
239
|
+
_nearest_match(value, candidates.map(&:to_s), max_distance)
|
|
240
|
+
end
|
|
223
241
|
hit && { value: hit[0], distance: hit[1] }
|
|
224
242
|
end
|
|
225
243
|
|
|
@@ -378,15 +396,18 @@ module Disarm
|
|
|
378
396
|
translate_errors { _strip_pua(text) }
|
|
379
397
|
end
|
|
380
398
|
|
|
381
|
-
# Strip "zalgo" combining-mark stacking, keeping at most `max_marks:`
|
|
382
|
-
# combining
|
|
383
|
-
|
|
399
|
+
# Strip "zalgo" combining-mark stacking, keeping at most `max_marks:` marks of each
|
|
400
|
+
# combining class on one base character. The default is the core's,
|
|
401
|
+
# DEFAULT_ZALGO_MAX_MARKS (3), equal to #zalgo?'s threshold (#788), so this never
|
|
402
|
+
# strips from text #zalgo? declines to flag.
|
|
403
|
+
def strip_zalgo(text, max_marks: DEFAULT_ZALGO_MAX_MARKS)
|
|
384
404
|
translate_errors { _strip_zalgo(text, max_marks) }
|
|
385
405
|
end
|
|
386
406
|
|
|
387
407
|
# Whether `text` looks like zalgo: any base character carries more than
|
|
388
|
-
# `threshold:`
|
|
389
|
-
|
|
408
|
+
# `threshold:` marks of one combining class. The default is the core's,
|
|
409
|
+
# DEFAULT_ZALGO_THRESHOLD (3).
|
|
410
|
+
def zalgo?(text, threshold: DEFAULT_ZALGO_THRESHOLD)
|
|
390
411
|
translate_errors { _zalgo?(text, threshold) }
|
|
391
412
|
end
|
|
392
413
|
|
|
@@ -422,13 +443,16 @@ module Disarm
|
|
|
422
443
|
# Turn arbitrary text into a safe filename. `platform:` is :universal
|
|
423
444
|
# (default), :windows, or :posix; `preserve_extension:` keeps the final
|
|
424
445
|
# extension when truncating to `max_length:`. Raises Disarm::InvalidArgument
|
|
425
|
-
# on an unknown platform
|
|
446
|
+
# on an unknown platform, or on a `separator:` that is not printable, non-space
|
|
447
|
+
# ASCII, contains a character illegal on the platform, or contains a path
|
|
448
|
+
# separator ("/" or "\\"). An empty separator is allowed, and so is " " on its
|
|
449
|
+
# own.
|
|
426
450
|
#
|
|
427
451
|
# A safe *filename*, not a safe URL path segment. "%" is legal in a filename, so one
|
|
428
452
|
# the caller typed is kept — sanitize_filename("..%2Fetc") returns "%2Fetc" — and a
|
|
429
453
|
# consumer that percent-decodes the result must validate AFTER decoding. What this
|
|
430
|
-
# will not do is manufacture one: "%"
|
|
431
|
-
#
|
|
454
|
+
# will not do is manufacture one: every "%" in the output is one the input
|
|
455
|
+
# contained, or part of the separator (#721).
|
|
432
456
|
def sanitize_filename(text, separator: "_", max_length: 255, platform: :universal,
|
|
433
457
|
lang: nil, preserve_extension: true)
|
|
434
458
|
translate_errors do
|
|
@@ -475,8 +499,9 @@ module Disarm
|
|
|
475
499
|
end
|
|
476
500
|
end
|
|
477
501
|
|
|
478
|
-
# ML/NLP normalization: NFKC → emoji→text → transliterate →
|
|
479
|
-
# [case fold] → strip control → strip zero-width →
|
|
502
|
+
# ML/NLP normalization: resolve deletions → NFKC → emoji→text → transliterate →
|
|
503
|
+
# strip accents → emoji→text → [case fold] → strip control → strip zero-width →
|
|
504
|
+
# collapse whitespace → NFC.
|
|
480
505
|
#
|
|
481
506
|
# `fold_case:` defaults to true. Pass false in front of a CASED model — folding is
|
|
482
507
|
# destructive, cannot be undone downstream, and an uncased evaluation harness cannot
|
|
@@ -513,7 +538,7 @@ module Disarm
|
|
|
513
538
|
# anomaly detector's `bidi` kind reports nine of the twelve, holding back LRM, RLM
|
|
514
539
|
# and ALM because a lone directional mark is ordinary in right-to-left text.
|
|
515
540
|
def bidi_control?(text)
|
|
516
|
-
_has_bidi_control?(text)
|
|
541
|
+
translate_errors { _has_bidi_control?(text) }
|
|
517
542
|
end
|
|
518
543
|
|
|
519
544
|
# Explain how `lang: "auto"` detection resolves `text`: a hash with
|
|
@@ -674,13 +699,14 @@ module Disarm
|
|
|
674
699
|
# whole surface. The original backtrace is preserved (passed as the third
|
|
675
700
|
# `raise` argument) so the failing native call site stays visible. A bad
|
|
676
701
|
# argument from the native layer can arrive as ArgumentError (an invalid
|
|
677
|
-
# scheme/target), TypeError (a non-String argument),
|
|
678
|
-
# negative max_length)
|
|
702
|
+
# scheme/target), TypeError (a non-String argument), RangeError (e.g. a
|
|
703
|
+
# negative max_length) or EncodingError (a String in an encoding Ruby cannot
|
|
704
|
+
# transcode to UTF-8) — all map to Disarm::InvalidArgument.
|
|
679
705
|
def translate_errors
|
|
680
706
|
yield
|
|
681
707
|
rescue Error
|
|
682
708
|
raise # already in our hierarchy — don't re-wrap
|
|
683
|
-
rescue ::ArgumentError, ::TypeError, ::RangeError => e
|
|
709
|
+
rescue ::ArgumentError, ::TypeError, ::RangeError, ::EncodingError => e
|
|
684
710
|
raise InvalidArgument, e.message, e.backtrace
|
|
685
711
|
rescue ::RuntimeError => e
|
|
686
712
|
raise Error, e.message, e.backtrace
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: disarm
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.17.2
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Richard Quinn
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-09-
|
|
11
|
+
date: 2026-09-26 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: rb_sys
|
|
@@ -130,7 +130,7 @@ required_ruby_version: !ruby/object:Gem::Requirement
|
|
|
130
130
|
requirements:
|
|
131
131
|
- - ">="
|
|
132
132
|
- !ruby/object:Gem::Version
|
|
133
|
-
version: 3.
|
|
133
|
+
version: 3.3.0
|
|
134
134
|
required_rubygems_version: !ruby/object:Gem::Requirement
|
|
135
135
|
requirements:
|
|
136
136
|
- - ">="
|