disarm 0.16.0 → 0.17.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 0b0ec44307d78ca53394c2b8aa21522f3e2e9a486c83bd536abdb8afd2b0ea7e
4
- data.tar.gz: e4e5edead5532bc809a8b2c3e4389b02177246d0d5c783bf5b24b641e32bb576
3
+ metadata.gz: 359f83c7f8f014a863c70caaa5bbea011530f5dc4866013bdd087ce6de77c7b4
4
+ data.tar.gz: 419a067a9dd9fde079844c8e5047b9be0b88cdb9a0f781a38c3f29a5e120ab10
5
5
  SHA512:
6
- metadata.gz: 65a1b31c791a89e407a697af6d89b8f65e772803705e0cb12d3a8b458d4bae093c1dfc558dfa9749db10857dcafe521b221d28979e22f1a37d66326cd2d86fe6
7
- data.tar.gz: a26f6cdfca9f10d678d5b8b56d243be22ac91a1d582d92a5fb6aee7d0f28b8abbe154da24e2bad1e924c1c5121f6c26baf937cba8f5caebc06cc954810d13458
6
+ metadata.gz: 9831ccfdd4b6fb49d8f5eb1df058226403fae7afc2ae933118f516fd1203a6e4dcd2c0d1dafd3ab5c63a081aa16013f95203c5df444a5bd8f1b6a5f479bb937a
7
+ data.tar.gz: a410e078c2b8092815673eaa244d1747a3dbff70e9ddcb789143c77f1617816715ea1dbe8f23fd1bf0231768c704da5c3dd7a197ed91c5af2fa3a7e157a6c741
data/README.md CHANGED
@@ -19,10 +19,10 @@ gem "disarm"
19
19
  gem install disarm
20
20
  ```
21
21
 
22
- Requires Ruby >= 3.1. Precompiled platform gems ship for Ruby 3.1 through 4.0
22
+ Requires Ruby >= 3.3. Precompiled platform gems ship for Ruby 3.3 through 4.0
23
23
  (Linux x86_64/aarch64, macOS x86_64/arm64, Windows). On a supported Ruby with no
24
24
  matching platform gem, the source gem installs and compiles locally, which needs a
25
- Rust toolchain. Below 3.1 the gem does not install at all.
25
+ Rust toolchain. Below 3.3 the gem does not install at all.
26
26
 
27
27
  ## Usage
28
28
 
@@ -13,7 +13,7 @@
13
13
  name = "disarm"
14
14
  version = "0.0.0"
15
15
  edition = "2021"
16
- rust-version = "1.81"
16
+ rust-version = "1.88"
17
17
  license = "MIT"
18
18
  description = "Ruby bindings for disarm — Unicode confusable/text-security building blocks"
19
19
  publish = false
@@ -30,7 +30,7 @@ crate-type = ["cdylib"]
30
30
  # dep — not a path — so the gem is self-contained and the rb-sys-dock cross-gem
31
31
  # build (which mounts only this gem dir) can fetch it. Imported as `disarm_core`
32
32
  # because the package name `disarm` would otherwise clash with this crate.
33
- disarm_core = { package = "disarm", version = "0.16", default-features = false }
33
+ disarm_core = { package = "disarm", version = "0.17", default-features = false }
34
34
  # magnus: ergonomic, safe Ruby<->Rust bindings over rb-sys. 0.8 supports Ruby >= 3.0
35
35
  # (the gem's required_ruby_version); rb-sys handles the platform glue.
36
36
  magnus = "0.8"
@@ -34,7 +34,8 @@
34
34
  use std::collections::HashSet;
35
35
 
36
36
  use disarm_core::api;
37
- use magnus::{function, method, prelude::*, Error, RHash, Ruby};
37
+ use magnus::encoding::EncodingCapable;
38
+ use magnus::{function, kwargs, method, prelude::*, Error, RHash, RString, Ruby};
38
39
 
39
40
  /// Map a `disarm` error onto the closest standard Ruby exception:
40
41
  /// `InvalidArgument` → `ArgumentError`, everything else → `RuntimeError`. The
@@ -147,9 +148,30 @@ fn wtf8_to_utf8(bytes: &[u8]) -> String {
147
148
  out
148
149
  }
149
150
 
150
- /// A text argument decoded at the boundary with the WTF-8 → UTF-8 contract (#472).
151
- /// Used in place of `String` for every text parameter; `Deref<Target = str>` lets the
152
- /// existing `&text` call sites reach the core unchanged.
151
+ /// A text argument decoded at the boundary by the encoding the String declares (#472,
152
+ /// and R1 of `formal/bindings`). Used in place of `String` for every text parameter;
153
+ /// `Deref<Target = str>` lets the existing `&text` call sites reach the core unchanged.
154
+ ///
155
+ /// The contract, per `String#encoding`:
156
+ ///
157
+ /// - **UTF-8**: the bytes as they are, with the WTF-8 scrub above for malformed input: a
158
+ /// well-formed surrogate pair recombines, each lone surrogate or undecodable byte is
159
+ /// one `U+FFFD`.
160
+ /// - **US-ASCII**: the bytes as they are; a byte above `0x7F` is not US-ASCII and is one
161
+ /// `U+FFFD`.
162
+ /// - **ASCII-8BIT (BINARY)**: bytes with no declared encoding, read as UTF-8 under the
163
+ /// same scrub as a UTF-8 String. `File.binread` and socket reads return BINARY for
164
+ /// what is usually UTF-8, and the C ABI reads its bytes the same way (#1020).
165
+ /// - **Any other encoding** (ISO-8859-1, Windows-1251, Shift_JIS, UTF-16LE, ...):
166
+ /// transcoded to UTF-8 by Ruby's own `String#encode`, each invalid or unmappable
167
+ /// sequence becoming one `U+FFFD`. An encoding Ruby cannot convert from at all (a
168
+ /// dummy such as UTF-7) raises `Encoding::ConverterNotFoundError`, which the Ruby
169
+ /// layer reports as `Disarm::InvalidArgument`.
170
+ ///
171
+ /// Until R1 every String's bytes were read as UTF-8 whatever it declared, so an
172
+ /// ISO-8859-1 `"caf\xE9"` transliterated to `"caf[?]"` and a Windows-1251 word to a row
173
+ /// of `[?]`, while the plain `String` parameters of the same functions, which magnus
174
+ /// converts, read the same bytes correctly.
153
175
  struct Wtf8Text(String);
154
176
 
155
177
  impl std::ops::Deref for Wtf8Text {
@@ -161,13 +183,43 @@ impl std::ops::Deref for Wtf8Text {
161
183
 
162
184
  impl magnus::TryConvert for Wtf8Text {
163
185
  fn try_convert(val: magnus::Value) -> Result<Self, Error> {
164
- let s = magnus::RString::try_convert(val)?;
186
+ let s = RString::try_convert(val)?;
187
+ // GVL invariant: argument conversion runs inside a Ruby method callback.
188
+ #[allow(clippy::expect_used)]
189
+ let ruby = Ruby::get().expect("argument conversion runs while holding the Ruby GVL");
190
+ let encoding = s.enc_get();
191
+ let usascii = encoding == ruby.usascii_encindex();
192
+ let s =
193
+ if encoding == ruby.utf8_encindex() || usascii || encoding == ruby.ascii8bit_encindex()
194
+ {
195
+ s
196
+ } else {
197
+ let options = kwargs!(
198
+ &ruby,
199
+ "invalid" => ruby.to_symbol("replace"),
200
+ "undef" => ruby.to_symbol("replace"),
201
+ "replace" => "\u{FFFD}"
202
+ );
203
+ s.funcall::<_, _, RString>("encode", (ruby.utf8_encoding(), options))?
204
+ };
165
205
  // SAFETY: the bytes are copied out immediately; no Ruby API runs between
166
206
  // `as_slice` and `to_vec`, so the string cannot be moved or collected.
167
207
  let bytes = unsafe { s.as_slice().to_vec() };
168
208
  let decoded = match std::str::from_utf8(&bytes) {
169
- Ok(valid) => valid.to_owned(),
170
- Err(_) => wtf8_to_utf8(&bytes),
209
+ Ok(valid) if !usascii || valid.is_ascii() => valid.to_owned(),
210
+ // A US-ASCII String carrying a byte above 0x7F: each such byte is invalid in
211
+ // the encoding the String declares, and is one U+FFFD.
212
+ _ if usascii => bytes
213
+ .iter()
214
+ .map(|&b| {
215
+ if b.is_ascii() {
216
+ char::from(b)
217
+ } else {
218
+ '\u{FFFD}'
219
+ }
220
+ })
221
+ .collect(),
222
+ _ => wtf8_to_utf8(&bytes),
171
223
  };
172
224
  Ok(Wtf8Text(decoded))
173
225
  }
@@ -191,6 +243,16 @@ fn transliterate_opts(
191
243
  scheme: String,
192
244
  lang: Option<String>,
193
245
  ) -> Result<String, Error> {
246
+ // `try_run` rejects an unknown `lang` (B2) and applies registered replacements.
247
+ transliterate_builder(&scheme, lang)?
248
+ .try_run(&text)
249
+ .map(std::borrow::Cow::into_owned)
250
+ .map_err(|e| map_err(&e))
251
+ }
252
+
253
+ /// The core's `Transliterate` builder for a scheme token and an optional `lang`; the
254
+ /// `lang` is checked when the builder runs.
255
+ fn transliterate_builder(scheme: &str, lang: Option<String>) -> Result<api::Transliterate, Error> {
194
256
  let mut builder = api::Transliterate::new();
195
257
  if scheme != "default" {
196
258
  let scheme: api::Scheme = scheme.parse().map_err(|e| map_err(&e))?;
@@ -199,7 +261,7 @@ fn transliterate_opts(
199
261
  if let Some(lang) = lang {
200
262
  builder = builder.lang(lang);
201
263
  }
202
- Ok(builder.run(&text).into_owned())
264
+ Ok(builder)
203
265
  }
204
266
 
205
267
  // ── Confusables (TR39) ────────────────────────────────────────────────────────
@@ -310,7 +372,7 @@ fn slugify(
310
372
  decimal: bool,
311
373
  hexadecimal: bool,
312
374
  safe_chars: String,
313
- ) -> String {
375
+ ) -> Result<String, Error> {
314
376
  let mut config = api::SlugConfig::default()
315
377
  .with_separator(separator)
316
378
  .with_lowercase(lowercase)
@@ -328,7 +390,8 @@ fn slugify(
328
390
  config.entities = entities;
329
391
  config.decimal = decimal;
330
392
  config.hexadecimal = hexadecimal;
331
- api::slugify(&text, &config)
393
+ // `try_slugify` rejects an unknown `lang` rather than falling back (B2).
394
+ api::try_slugify(&text, &config).map_err(|e| map_err(&e))
332
395
  }
333
396
 
334
397
  /// `Disarm._replace_emoji(text, replacement)` — every emoji replaced, verbatim (#972).
@@ -336,7 +399,8 @@ fn replace_emoji(text: Wtf8Text, replacement: String) -> String {
336
399
  api::replace_emoji(&text, &replacement)
337
400
  }
338
401
 
339
- /// `Disarm._demojize(text, strip_modifiers)`.
402
+ /// `Disarm._demojize(text, strip_modifiers)`. An emoji CLDR cannot name becomes `[?]`,
403
+ /// as in every binding.
340
404
  fn demojize(text: Wtf8Text, strip_modifiers: bool) -> String {
341
405
  api::demojize(&text, strip_modifiers)
342
406
  }
@@ -401,9 +465,14 @@ fn catalog_key(
401
465
  strict_iso9: bool,
402
466
  digit_policy: String,
403
467
  ) -> Result<String, Error> {
404
- api::catalog_key_with(&text, lang.as_deref(), strict_iso9, parse_policy(&digit_policy)?)
405
- .map(std::borrow::Cow::into_owned)
406
- .map_err(|e| map_err(&e))
468
+ api::catalog_key_with(
469
+ &text,
470
+ lang.as_deref(),
471
+ strict_iso9,
472
+ parse_policy(&digit_policy)?,
473
+ )
474
+ .map(std::borrow::Cow::into_owned)
475
+ .map_err(|e| map_err(&e))
407
476
  }
408
477
 
409
478
  /// `Disarm._skeleton_key(text, digit_policy)` — the TR39 identifier skeleton plus the two
@@ -628,16 +697,9 @@ fn find_untranslatable(
628
697
  scheme: String,
629
698
  lang: Option<String>,
630
699
  ) -> Result<Vec<(String, usize)>, Error> {
631
- let mut builder = api::Transliterate::new();
632
- if scheme != "default" {
633
- let scheme: api::Scheme = scheme.parse().map_err(|e| map_err(&e))?;
634
- builder = builder.scheme(scheme);
635
- }
636
- if let Some(lang) = lang {
637
- builder = builder.lang(lang);
638
- }
639
- Ok(builder
640
- .find_untranslatable(&text)
700
+ Ok(transliterate_builder(&scheme, lang)?
701
+ .try_find_untranslatable(&text)
702
+ .map_err(|e| map_err(&e))?
641
703
  .into_iter()
642
704
  .map(|u| (u.ch.to_string(), u.offset))
643
705
  .collect())
@@ -918,6 +980,12 @@ fn get_pipeline(profile: String) -> Result<Pipeline, Error> {
918
980
  fn init(ruby: &Ruby) -> Result<(), Error> {
919
981
  let module = ruby.define_module("Disarm")?;
920
982
 
983
+ // The zalgo defaults, read from the core so the Ruby layer never restates them: it
984
+ // kept a literal cap of 2 after #788 raised the core's to 3 (`formal/bindings`, B1).
985
+ // Defined before `lib/disarm.rb`'s method definitions read them as defaults.
986
+ module.const_set("DEFAULT_ZALGO_THRESHOLD", api::DEFAULT_ZALGO_THRESHOLD)?;
987
+ module.const_set("DEFAULT_ZALGO_MAX_MARKS", api::DEFAULT_ZALGO_MAX_MARKS)?;
988
+
921
989
  // Raw, `_`-prefixed shims wrapped by the idiomatic Ruby layer (#357).
922
990
  module.define_singleton_method("_transliterate", function!(transliterate, 1))?;
923
991
  module.define_singleton_method("_transliterate_opts", function!(transliterate_opts, 3))?;
@@ -948,10 +1016,7 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
948
1016
  // keeping `rescue Disarm::Error` exhaustive across the whole public surface.
949
1017
  module.define_singleton_method("_strip_accents", function!(strip_accents, 1))?;
950
1018
  module.define_singleton_method("_fold_case", function!(fold_case, 1))?;
951
- module.define_singleton_method(
952
- "_is_case_fold_stable?",
953
- function!(is_case_fold_stable, 1),
954
- )?;
1019
+ module.define_singleton_method("_is_case_fold_stable?", function!(is_case_fold_stable, 1))?;
955
1020
  module.define_singleton_method("_find_key_collisions", function!(find_key_collisions, 3))?;
956
1021
  module.define_singleton_method("_suspicious_hostname?", function!(suspicious_hostname, 1))?;
957
1022
  module.define_singleton_method("_analyze_hostname", function!(analyze_hostname, 2))?;
@@ -990,10 +1055,7 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
990
1055
  function!(reverse_transliterate, 2),
991
1056
  )?;
992
1057
  module.define_singleton_method("_find_untranslatable", function!(find_untranslatable, 3))?;
993
- module.define_singleton_method(
994
- "_unmapped_confusables",
995
- function!(unmapped_confusables, 1),
996
- )?;
1058
+ module.define_singleton_method("_unmapped_confusables", function!(unmapped_confusables, 1))?;
997
1059
  module.define_singleton_method(
998
1060
  "_find_unmapped_confusables",
999
1061
  function!(find_unmapped_confusables, 2),
@@ -1008,19 +1070,10 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
1008
1070
  // Metadata introspection (#404 phase 3 parity backfill).
1009
1071
  module.define_singleton_method("_lang_info", function!(lang_info, 1))?;
1010
1072
  module.define_singleton_method("_script_info", function!(script_info, 1))?;
1011
- module.define_singleton_method(
1012
- "_confusable_coverage",
1013
- function!(confusable_coverage, 1),
1014
- )?;
1015
- module.define_singleton_method(
1016
- "_confusables_version",
1017
- function!(confusables_version, 0),
1018
- )?;
1073
+ module.define_singleton_method("_confusable_coverage", function!(confusable_coverage, 1))?;
1074
+ module.define_singleton_method("_confusables_version", function!(confusables_version, 0))?;
1019
1075
  module.define_singleton_method("_unicode_version", function!(unicode_version, 0))?;
1020
- module.define_singleton_method(
1021
- "_key_schema_version",
1022
- function!(key_schema_version, 0),
1023
- )?;
1076
+ module.define_singleton_method("_key_schema_version", function!(key_schema_version, 0))?;
1024
1077
  module.define_singleton_method("_list_scripts", function!(list_scripts, 0))?;
1025
1078
  module.define_singleton_method("_list_context_langs", function!(list_context_langs, 0))?;
1026
1079
 
@@ -2,5 +2,5 @@
2
2
 
3
3
  module Disarm
4
4
  # Kept in lockstep with the Rust crate / Python package version.
5
- VERSION = "0.16.0"
5
+ VERSION = "0.17.2"
6
6
  end
data/lib/disarm.rb CHANGED
@@ -24,6 +24,13 @@ end
24
24
  # keyword arguments with the core's defaults, symbol tokens (:latin, :default, …),
25
25
  # a single transliterate(text, scheme:) entrypoint, and a Disarm::Error hierarchy.
26
26
  # Each method is still a thin wrapper over the pure-Rust `disarm` core.
27
+ #
28
+ # Text arguments are read by the encoding the String declares. UTF-8 and US-ASCII are
29
+ # read as they are, and ASCII-8BIT (BINARY) is read as UTF-8, the way the C ABI reads
30
+ # its bytes; any other encoding (ISO-8859-1, Windows-1251, UTF-16LE, ...) is transcoded
31
+ # to UTF-8 first. A malformed or unmappable sequence becomes one U+FFFD, and so does a
32
+ # lone surrogate, the contract every binding shares (#469). An encoding Ruby cannot
33
+ # convert from at all raises Disarm::InvalidArgument.
27
34
  module Disarm
28
35
  # Base class for every error disarm raises, so consumers can `rescue
29
36
  # Disarm::Error`. The native shim raises Ruby's built-in ArgumentError /
@@ -38,7 +45,8 @@ module Disarm
38
45
  # Transliterate Unicode text to ASCII. `scheme:` selects the standard:
39
46
  # :default (the general-purpose scheme), :strict_iso9, or :gost7034. `lang:`
40
47
  # applies a language profile on top of the scheme (e.g. "uk" → Київ → "Kyiv",
41
- # "de" → ü → "ue"); nil means no profile. Both accept a String or Symbol.
48
+ # "de" → ü → "ue"); nil means no profile. Both accept a String or Symbol. Raises
49
+ # Disarm::InvalidArgument on an unknown lang, as every binding does.
42
50
  def transliterate(text, scheme: :default, lang: nil)
43
51
  scheme = scheme.to_s
44
52
  lang = lang&.to_s
@@ -53,7 +61,8 @@ module Disarm
53
61
  end
54
62
  end
55
63
 
56
- # Fold cross-script confusables toward `target:` (:latin or :cyrillic).
64
+ # Fold cross-script confusables toward `target:` (:latin, :cyrillic, :arabic or
65
+ # :hebrew).
57
66
  #
58
67
  # `digit_policy:` selects how non-Latin DIGITS fold (#561).
59
68
  #
@@ -61,9 +70,9 @@ module Disarm
61
70
  # right for prose, where a Devanagari zero really is a zero.
62
71
  #
63
72
  # `:tr39` uses upstream's targets, which send most of them to a Latin letter
64
- # (`०` → `o`; three of the 45 rows fold to `.` or to the two characters `rn`
73
+ # (`०` → `o`; three of the 47 rows fold to `.` or to the two characters `rn`
65
74
  # instead). That is what an identifier *skeleton* wants, since its only job is to
66
- # make two confusable identifiers collide. The two differ on 45 rows and agree
75
+ # make two confusable identifiers collide. The two differ on 47 rows and agree
67
76
  # everywhere else. Scoped to `target: :latin` — the override rows are generated from
68
77
  # the Latin table and carry TR39's Latin-script targets, so with `target: :cyrillic`
69
78
  # it is a no-op.
@@ -75,15 +84,16 @@ module Disarm
75
84
  translate_errors { _normalize_confusables(text, target.to_s, digit_policy.to_s) }
76
85
  end
77
86
 
78
- # Whether `text` contains a character confusable with `target:` (:latin or
79
- # :cyrillic).
87
+ # Whether `text` contains a character confusable with `target:` (:latin,
88
+ # :cyrillic, :arabic or :hebrew).
80
89
  def confusable?(text, target: :latin)
81
90
  translate_errors { _confusable?(text, target.to_s) }
82
91
  end
83
92
 
84
93
  # Generate a URL-safe slug. Mirrors the core's `SlugConfig` defaults; every
85
94
  # option past `text` is keyword-only. (`regex_pattern`/`replacements` are not
86
- # surfaced yet — see ext/disarm/src/lib.rs.)
95
+ # surfaced yet — see ext/disarm/src/lib.rs.) Raises Disarm::InvalidArgument on an
96
+ # unknown `lang:`.
87
97
  def slugify(
88
98
  text,
89
99
  separator: "-",
@@ -111,7 +121,9 @@ module Disarm
111
121
  end
112
122
 
113
123
  # Replace emoji with their plain names (e.g. "👍" → "thumbs up").
114
- # `strip_modifiers:` drops skin-tone / variation modifiers before naming.
124
+ # `strip_modifiers:` drops skin-tone / variation modifiers before naming. An emoji
125
+ # CLDR cannot name (a regional indicator or a Plane 14 tag character standing
126
+ # alone) becomes "[?]", the sentinel #transliterate writes, as in every binding.
115
127
  def demojize(text, strip_modifiers: false)
116
128
  translate_errors { _demojize(text, strip_modifiers) }
117
129
  end
@@ -122,7 +134,8 @@ module Disarm
122
134
  # +demojize+ asks what CLDR calls a character, so its domain is the name table, which
123
135
  # is wider than the emoji: <tt>demojize("x\u2122y")</tt> is "x trade mark y". This
124
136
  # asks whether the UCD calls it an emoji — Emoji_Presentation=Yes, an Emoji=Yes base
125
- # carrying U+FE0F, and the ZWJ, modifier, keycap and flag sequences on those. Nothing
137
+ # carrying U+FE0F (not an Extended_Pictographic one, so a black star with a selector
138
+ # stays), and the ZWJ, modifier, keycap and flag sequences on those. Nothing
126
139
  # else moves.
127
140
  #
128
141
  # +replacement+ is inserted exactly as given: "" closes an intra-word split and " "
@@ -146,7 +159,7 @@ module Disarm
146
159
  # keeps the VS15/VS16 presentation selectors after a base, which the naive chain
147
160
  # deletes, and it collapses TAB/LF to a space, which the primitives leave alone.
148
161
  def strip_format(text)
149
- _strip_format(text)
162
+ translate_errors { _strip_format(text) }
150
163
  end
151
164
 
152
165
  # Remove obfuscation (zero-width, bidi, combining-mark abuse) while keeping
@@ -219,7 +232,12 @@ module Disarm
219
232
  # `max_distance:` (#894). Reports; it does not decide. An exact match is reported with
220
233
  # distance 0, and ties go to the first candidate at the lowest distance.
221
234
  def nearest_match(value, candidates, max_distance: 1)
222
- hit = translate_errors { _nearest_match(value, candidates.map(&:to_s), max_distance) }
235
+ hit = translate_errors do
236
+ # A non-collection is a wrong-typed argument, not a NoMethodError (N2).
237
+ raise ::TypeError, "candidates must be an Array or Enumerable" unless candidates.respond_to?(:map)
238
+
239
+ _nearest_match(value, candidates.map(&:to_s), max_distance)
240
+ end
223
241
  hit && { value: hit[0], distance: hit[1] }
224
242
  end
225
243
 
@@ -378,15 +396,18 @@ module Disarm
378
396
  translate_errors { _strip_pua(text) }
379
397
  end
380
398
 
381
- # Strip "zalgo" combining-mark stacking, keeping at most `max_marks:` (2)
382
- # combining marks per base character.
383
- def strip_zalgo(text, max_marks: 2)
399
+ # Strip "zalgo" combining-mark stacking, keeping at most `max_marks:` marks of each
400
+ # combining class on one base character. The default is the core's,
401
+ # DEFAULT_ZALGO_MAX_MARKS (3), equal to #zalgo?'s threshold (#788), so this never
402
+ # strips from text #zalgo? declines to flag.
403
+ def strip_zalgo(text, max_marks: DEFAULT_ZALGO_MAX_MARKS)
384
404
  translate_errors { _strip_zalgo(text, max_marks) }
385
405
  end
386
406
 
387
407
  # Whether `text` looks like zalgo: any base character carries more than
388
- # `threshold:` (3) combining marks.
389
- def zalgo?(text, threshold: 3)
408
+ # `threshold:` marks of one combining class. The default is the core's,
409
+ # DEFAULT_ZALGO_THRESHOLD (3).
410
+ def zalgo?(text, threshold: DEFAULT_ZALGO_THRESHOLD)
390
411
  translate_errors { _zalgo?(text, threshold) }
391
412
  end
392
413
 
@@ -422,13 +443,16 @@ module Disarm
422
443
  # Turn arbitrary text into a safe filename. `platform:` is :universal
423
444
  # (default), :windows, or :posix; `preserve_extension:` keeps the final
424
445
  # extension when truncating to `max_length:`. Raises Disarm::InvalidArgument
425
- # on an unknown platform.
446
+ # on an unknown platform, or on a `separator:` that is not printable, non-space
447
+ # ASCII, contains a character illegal on the platform, or contains a path
448
+ # separator ("/" or "\\"). An empty separator is allowed, and so is " " on its
449
+ # own.
426
450
  #
427
451
  # A safe *filename*, not a safe URL path segment. "%" is legal in a filename, so one
428
452
  # the caller typed is kept — sanitize_filename("..%2Fetc") returns "%2Fetc" — and a
429
453
  # consumer that percent-decodes the result must validate AFTER decoding. What this
430
- # will not do is manufacture one: "%" never appears in the output unless it appeared
431
- # in the input (#721).
454
+ # will not do is manufacture one: every "%" in the output is one the input
455
+ # contained, or part of the separator (#721).
432
456
  def sanitize_filename(text, separator: "_", max_length: 255, platform: :universal,
433
457
  lang: nil, preserve_extension: true)
434
458
  translate_errors do
@@ -475,8 +499,9 @@ module Disarm
475
499
  end
476
500
  end
477
501
 
478
- # ML/NLP normalization: NFKC → emoji→text → transliterate → strip accents →
479
- # [case fold] → strip control → strip zero-width → collapse whitespace.
502
+ # ML/NLP normalization: resolve deletions → NFKC → emoji→text → transliterate →
503
+ # strip accents → emoji→text → [case fold] → strip control → strip zero-width →
504
+ # collapse whitespace → NFC.
480
505
  #
481
506
  # `fold_case:` defaults to true. Pass false in front of a CASED model — folding is
482
507
  # destructive, cannot be undone downstream, and an uncased evaluation harness cannot
@@ -513,7 +538,7 @@ module Disarm
513
538
  # anomaly detector's `bidi` kind reports nine of the twelve, holding back LRM, RLM
514
539
  # and ALM because a lone directional mark is ordinary in right-to-left text.
515
540
  def bidi_control?(text)
516
- _has_bidi_control?(text)
541
+ translate_errors { _has_bidi_control?(text) }
517
542
  end
518
543
 
519
544
  # Explain how `lang: "auto"` detection resolves `text`: a hash with
@@ -674,13 +699,14 @@ module Disarm
674
699
  # whole surface. The original backtrace is preserved (passed as the third
675
700
  # `raise` argument) so the failing native call site stays visible. A bad
676
701
  # argument from the native layer can arrive as ArgumentError (an invalid
677
- # scheme/target), TypeError (a non-String argument), or RangeError (e.g. a
678
- # negative max_length) — all map to Disarm::InvalidArgument.
702
+ # scheme/target), TypeError (a non-String argument), RangeError (e.g. a
703
+ # negative max_length) or EncodingError (a String in an encoding Ruby cannot
704
+ # transcode to UTF-8) — all map to Disarm::InvalidArgument.
679
705
  def translate_errors
680
706
  yield
681
707
  rescue Error
682
708
  raise # already in our hierarchy — don't re-wrap
683
- rescue ::ArgumentError, ::TypeError, ::RangeError => e
709
+ rescue ::ArgumentError, ::TypeError, ::RangeError, ::EncodingError => e
684
710
  raise InvalidArgument, e.message, e.backtrace
685
711
  rescue ::RuntimeError => e
686
712
  raise Error, e.message, e.backtrace
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: disarm
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.16.0
4
+ version: 0.17.2
5
5
  platform: ruby
6
6
  authors:
7
7
  - Richard Quinn
8
8
  autorequire:
9
9
  bindir: bin
10
10
  cert_chain: []
11
- date: 2026-09-06 00:00:00.000000000 Z
11
+ date: 2026-09-26 00:00:00.000000000 Z
12
12
  dependencies:
13
13
  - !ruby/object:Gem::Dependency
14
14
  name: rb_sys
@@ -130,7 +130,7 @@ required_ruby_version: !ruby/object:Gem::Requirement
130
130
  requirements:
131
131
  - - ">="
132
132
  - !ruby/object:Gem::Version
133
- version: 3.1.0
133
+ version: 3.3.0
134
134
  required_rubygems_version: !ruby/object:Gem::Requirement
135
135
  requirements:
136
136
  - - ">="