faster_path 0.3.10 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/Cargo.lock +19 -37
- data/Cargo.toml +5 -6
- data/Gemfile +3 -7
- data/README.md +94 -30
- data/Rakefile +11 -2
- data/faster_path.gemspec +6 -3
- data/lib/faster_path/version.rb +1 -1
- data/lib/faster_path.rb +4 -2
- data/src/basename.rs +166 -69
- data/src/chop_basename.rs +84 -21
- data/src/cleanpath_aggressive.rs +87 -71
- data/src/cleanpath_conservative.rs +175 -79
- data/src/dirname.rs +96 -32
- data/src/extname.rs +150 -20
- data/src/lib.rs +54 -104
- data/src/path_parsing.rs +282 -21
- data/src/pathname.rs +296 -256
- data/src/plus.rs +90 -74
- data/src/prepend_prefix.rs +11 -13
- data/src/relative_path_from.rs +72 -48
- data/src/ruby.rs +530 -0
- metadata +17 -18
- data/src/debug.rs +0 -29
- data/src/helpers.rs +0 -51
- data/src/memrnchr.rs +0 -70
- data/src/pathname_sys.rs +0 -41
data/src/ruby.rs
ADDED
|
@@ -0,0 +1,530 @@
|
|
|
1
|
+
// Glue between Ruby and the path functions, written so that nothing Ruby
|
|
2
|
+
// passes in can crash the interpreter:
|
|
3
|
+
//
|
|
4
|
+
// * Panics never unwind into Ruby; they are caught and raised as `RuntimeError`.
|
|
5
|
+
// * Ruby exceptions are only raised once every Rust value owning memory has
|
|
6
|
+
// been dropped, since raising `longjmp`s past Rust frames without running
|
|
7
|
+
// their destructors.
|
|
8
|
+
// * Ruby code (`to_path`, `to_s`, ...) is only called through `protect_send`,
|
|
9
|
+
// so an exception it raises comes back as an `Err` instead of jumping over
|
|
10
|
+
// Rust frames.
|
|
11
|
+
// * Strings are read as bytes in their own encoding, never assumed to be UTF-8.
|
|
12
|
+
use std::any::Any;
|
|
13
|
+
use std::borrow::Cow;
|
|
14
|
+
use std::io;
|
|
15
|
+
use std::path::Path;
|
|
16
|
+
use std::sync::atomic::{AtomicI32, AtomicU8, Ordering};
|
|
17
|
+
use std::thread;
|
|
18
|
+
|
|
19
|
+
use rutie::rubysys;
|
|
20
|
+
use rutie::types::{c_char, c_long, Argc, EncodingIndex, Value, ValueType};
|
|
21
|
+
use rutie::{AnyException, AnyObject, Class, Encoding, NilClass, Object, RString, Symbol, VM};
|
|
22
|
+
|
|
23
|
+
use crate::path_parsing::Rules;
|
|
24
|
+
|
|
25
|
+
pub type RubyResult = Result<AnyObject, AnyException>;
|
|
26
|
+
|
|
27
|
+
// Defines Ruby methods taking any number of arguments (arity -1). The body
|
|
28
|
+
// sees the arguments as `&[AnyObject]` and returns a `RubyResult`; an `Err`
|
|
29
|
+
// is raised in Ruby.
|
|
30
|
+
macro_rules! ruby_methods {
|
|
31
|
+
($($(#[$attribute:meta])* fn $name:ident($arguments:ident) $body:block)*) => {$(
|
|
32
|
+
rutie::rutie_callback! {
|
|
33
|
+
$(#[$attribute])*
|
|
34
|
+
pub fn $name(argc: rutie::types::Argc, argv: *const rutie::AnyObject, _itself: rutie::AnyObject) -> rutie::AnyObject {
|
|
35
|
+
// Safety: Ruby passes `argc` arguments in `argv`, alive for the whole call.
|
|
36
|
+
let $arguments = unsafe { $crate::ruby::arguments(argc, argv) };
|
|
37
|
+
$crate::ruby::finish(std::panic::catch_unwind(std::panic::AssertUnwindSafe(
|
|
38
|
+
|| -> $crate::ruby::RubyResult { $body }
|
|
39
|
+
)))
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
)*};
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/// Borrows the arguments of a method call.
|
|
46
|
+
///
|
|
47
|
+
/// # Safety
|
|
48
|
+
///
|
|
49
|
+
/// `argv` must point to `argc` Ruby values that stay alive for `'a`, as the
|
|
50
|
+
/// arguments of a method call do while the method runs.
|
|
51
|
+
pub unsafe fn arguments<'a>(argc: Argc, argv: *const AnyObject) -> &'a [AnyObject] {
|
|
52
|
+
if argc <= 0 || argv.is_null() {
|
|
53
|
+
&[]
|
|
54
|
+
} else {
|
|
55
|
+
// Safety: guaranteed by the caller.
|
|
56
|
+
unsafe { std::slice::from_raw_parts(argv, argc as usize) }
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/// Returns the method's result to Ruby, or raises its error.
|
|
61
|
+
///
|
|
62
|
+
/// Raising does not return, so it happens here, after the method body and
|
|
63
|
+
/// everything it allocated have been dropped.
|
|
64
|
+
pub fn finish(result: thread::Result<RubyResult>) -> AnyObject {
|
|
65
|
+
let exception = match result {
|
|
66
|
+
Ok(Ok(value)) => return value,
|
|
67
|
+
Ok(Err(exception)) => exception,
|
|
68
|
+
Err(payload) => panic_exception(payload),
|
|
69
|
+
};
|
|
70
|
+
VM::raise_ex(exception);
|
|
71
|
+
NilClass::new().into()
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
fn panic_exception(payload: Box<dyn Any + Send>) -> AnyException {
|
|
75
|
+
let message = if let Some(message) = payload.downcast_ref::<&str>() {
|
|
76
|
+
message.to_string()
|
|
77
|
+
} else if let Some(message) = payload.downcast_ref::<String>() {
|
|
78
|
+
message.clone()
|
|
79
|
+
} else {
|
|
80
|
+
"unknown error".to_string()
|
|
81
|
+
};
|
|
82
|
+
error(&Class::runtime_error(), &format!("faster_path internal error: {}", message))
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
pub fn error(class: &Class, message: &str) -> AnyException {
|
|
86
|
+
AnyException::from_class(class, message)
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
pub fn argument_error(message: &str) -> AnyException {
|
|
90
|
+
error(&Class::argument_error(), message)
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/// Raises `ArgumentError` unless there are at most `max` arguments.
|
|
94
|
+
pub fn check_max_arguments(arguments: &[AnyObject], max: usize) -> Result<(), AnyException> {
|
|
95
|
+
if arguments.len() > max {
|
|
96
|
+
let expected = if max == 0 { "0".to_string() } else { format!("0..{}", max) };
|
|
97
|
+
Err(argument_error(&format!(
|
|
98
|
+
"wrong number of arguments (given {}, expected {})",
|
|
99
|
+
arguments.len(),
|
|
100
|
+
expected
|
|
101
|
+
)))
|
|
102
|
+
} else {
|
|
103
|
+
Ok(())
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/// Truthiness of an optional argument.
|
|
108
|
+
pub fn truthy_argument(arguments: &[AnyObject], index: usize, default: bool) -> bool {
|
|
109
|
+
match arguments.get(index) {
|
|
110
|
+
Some(argument) => !(argument.is_nil() || argument.value().is_false()),
|
|
111
|
+
None => default,
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/// A path argument.
|
|
116
|
+
pub enum PathArgument {
|
|
117
|
+
/// Missing, `nil`, or neither a `String` nor something with `to_path`.
|
|
118
|
+
/// The methods have always treated these as an empty path.
|
|
119
|
+
Missing,
|
|
120
|
+
/// A `String` argument.
|
|
121
|
+
String(RString),
|
|
122
|
+
/// The `String` an argument's `to_path` returned.
|
|
123
|
+
Converted(RString),
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
impl PathArgument {
|
|
127
|
+
/// Reads the argument at `index` like Ruby's `File` methods do: a `String`,
|
|
128
|
+
/// or an object with `to_path`.
|
|
129
|
+
pub fn new(arguments: &[AnyObject], index: usize) -> Result<Self, AnyException> {
|
|
130
|
+
let argument = match arguments.get(index) {
|
|
131
|
+
Some(argument) => argument,
|
|
132
|
+
None => return Ok(PathArgument::Missing),
|
|
133
|
+
};
|
|
134
|
+
if let Ok(string) = argument.try_convert_to::<RString>() {
|
|
135
|
+
return Ok(PathArgument::String(string));
|
|
136
|
+
}
|
|
137
|
+
if argument.is_nil() {
|
|
138
|
+
return Ok(PathArgument::Missing);
|
|
139
|
+
}
|
|
140
|
+
let responds_to_path = argument.protect_send("respond_to?", &[Symbol::new("to_path").into()])?;
|
|
141
|
+
if !responds_to_path.value().is_true() {
|
|
142
|
+
return Ok(PathArgument::Missing);
|
|
143
|
+
}
|
|
144
|
+
let path = argument.protect_send("to_path", &[])?;
|
|
145
|
+
Ok(PathArgument::Converted(expect_string(path, argument, "to_path")?))
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
pub fn is_missing(&self) -> bool {
|
|
149
|
+
matches!(self, PathArgument::Missing)
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/// The path, or `default` if the argument is missing.
|
|
153
|
+
pub fn path_or(&self, default: &'static [u8]) -> Result<PathString<'_>, AnyException> {
|
|
154
|
+
match self {
|
|
155
|
+
PathArgument::Missing => Ok(PathString::literal(default)),
|
|
156
|
+
PathArgument::String(string) => PathString::new(string),
|
|
157
|
+
// Nothing but this value refers to the string `to_path` returned, and
|
|
158
|
+
// the garbage collector may not see it, so keep a copy.
|
|
159
|
+
PathArgument::Converted(string) => PathString::new(string).map(PathString::into_owned),
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
pub fn path(&self) -> Result<PathString<'_>, AnyException> {
|
|
164
|
+
self.path_or(b"")
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/// The `String`, for error messages or to return unchanged.
|
|
168
|
+
pub fn string(&self) -> Option<&RString> {
|
|
169
|
+
match self {
|
|
170
|
+
PathArgument::Missing => None,
|
|
171
|
+
PathArgument::String(string) | PathArgument::Converted(string) => Some(string),
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/// A path read from a Ruby `String`: its bytes and its encoding.
|
|
177
|
+
pub struct PathString<'a> {
|
|
178
|
+
original: Cow<'a, [u8]>,
|
|
179
|
+
// On Windows, a copy in which each `\` that is part of a multibyte
|
|
180
|
+
// character is a NUL instead, so it isn't taken for a separator; see
|
|
181
|
+
// `EncodingId::mask_backslashes`.
|
|
182
|
+
masked: Option<Vec<u8>>,
|
|
183
|
+
pub encoding: EncodingId,
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
impl<'a> PathString<'a> {
|
|
187
|
+
/// An empty path, used where a `String` argument is missing.
|
|
188
|
+
pub fn empty() -> Self {
|
|
189
|
+
PathString::literal(b"")
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
pub fn literal(bytes: &'static [u8]) -> Self {
|
|
193
|
+
PathString { original: Cow::Borrowed(bytes), masked: None, encoding: EncodingId::us_ascii() }
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
/// Reads `string`, which must be kept alive (and unmodified) while the
|
|
197
|
+
/// result is in use; a method argument is.
|
|
198
|
+
///
|
|
199
|
+
/// Like Ruby's own path methods this rejects strings that are not
|
|
200
|
+
/// ASCII-compatible (such as UTF-16), since those can't be split on `/`,
|
|
201
|
+
/// and strings containing a null byte.
|
|
202
|
+
pub fn new(string: &'a RString) -> Result<Self, AnyException> {
|
|
203
|
+
let encoding = EncodingId::of(string);
|
|
204
|
+
if !encoding.is_ascii_compatible() {
|
|
205
|
+
return Err(error(
|
|
206
|
+
&Class::encoding_compatibility_error(),
|
|
207
|
+
&format!("path name must be ASCII-compatible ({}): {}", encoding.name(), inspect(string)?),
|
|
208
|
+
));
|
|
209
|
+
}
|
|
210
|
+
// This returns the bytes of a `String`; `RString` is always one.
|
|
211
|
+
let bytes = string.to_bytes_unchecked();
|
|
212
|
+
if bytes.contains(&0) {
|
|
213
|
+
return Err(argument_error("path name contains null byte"));
|
|
214
|
+
}
|
|
215
|
+
let masked = if Rules::NATIVE.is_dosish() { encoding.mask_backslashes(bytes) } else { None };
|
|
216
|
+
Ok(PathString { original: Cow::Borrowed(bytes), masked, encoding })
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
pub fn into_owned(self) -> PathString<'static> {
|
|
220
|
+
PathString { original: Cow::Owned(self.original.into_owned()), masked: self.masked, encoding: self.encoding }
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/// The bytes the path functions work on.
|
|
224
|
+
pub fn bytes(&self) -> &[u8] {
|
|
225
|
+
self.masked.as_deref().unwrap_or(&self.original)
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/// A new Ruby string with `bytes` in this path's encoding.
|
|
229
|
+
pub fn to_ruby(&self, bytes: &[u8]) -> RString {
|
|
230
|
+
self.encoding.new_string(bytes)
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/// Whether `pos` starts a character of `bytes`, which must start at a
|
|
234
|
+
/// character of this path's encoding.
|
|
235
|
+
pub fn is_char_boundary(&self, bytes: &[u8], pos: usize) -> bool {
|
|
236
|
+
if pos == 0 || pos >= bytes.len() {
|
|
237
|
+
return true;
|
|
238
|
+
}
|
|
239
|
+
if self.encoding == EncodingId::utf8() {
|
|
240
|
+
// Not a continuation byte
|
|
241
|
+
return (bytes[pos] & 0xC0) != 0x80;
|
|
242
|
+
}
|
|
243
|
+
if self.encoding.is_single_byte_or_utf8() {
|
|
244
|
+
return true;
|
|
245
|
+
}
|
|
246
|
+
// Other encodings: walk the characters from the start like Ruby does.
|
|
247
|
+
let bytes = unmask(bytes);
|
|
248
|
+
let mut index = 0;
|
|
249
|
+
while index < pos {
|
|
250
|
+
index += self.encoding.char_len(&bytes, index);
|
|
251
|
+
}
|
|
252
|
+
index == pos
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
/// The path to give the OS.
|
|
256
|
+
#[cfg(unix)]
|
|
257
|
+
pub fn os_path(&self) -> io::Result<Cow<'_, Path>> {
|
|
258
|
+
use std::os::unix::ffi::OsStrExt;
|
|
259
|
+
Ok(Cow::Borrowed(Path::new(std::ffi::OsStr::from_bytes(&self.original))))
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/// The path to give the OS, which takes Unicode on Windows.
|
|
263
|
+
#[cfg(not(unix))]
|
|
264
|
+
pub fn os_path(&self) -> io::Result<Cow<'_, Path>> {
|
|
265
|
+
let invalid = || io::Error::new(io::ErrorKind::InvalidInput, "path name can't be converted to UTF-8");
|
|
266
|
+
if self.encoding == EncodingId::utf8() || self.original.is_ascii() {
|
|
267
|
+
return std::str::from_utf8(&self.original).map(|path| Cow::Borrowed(Path::new(path))).map_err(|_| invalid());
|
|
268
|
+
}
|
|
269
|
+
let utf8 = self.encoding.transcode(&self.original, EncodingId::utf8()).ok_or_else(invalid)?;
|
|
270
|
+
String::from_utf8(utf8).map(|path| Cow::Owned(path.into())).map_err(|_| invalid())
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
// Puts back the `\`s `EncodingId::mask_backslashes` replaced. NUL is never
|
|
275
|
+
// in a path otherwise.
|
|
276
|
+
fn unmask(bytes: &[u8]) -> Cow<'_, [u8]> {
|
|
277
|
+
if Rules::NATIVE.is_dosish() && memchr::memchr(0, bytes).is_some() {
|
|
278
|
+
Cow::Owned(bytes.iter().map(|&c| if c == 0 { b'\\' } else { c }).collect())
|
|
279
|
+
} else {
|
|
280
|
+
Cow::Borrowed(bytes)
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
/// An encoding, by its index in Ruby's table of encodings.
|
|
285
|
+
///
|
|
286
|
+
/// `Encoding` objects compare with `Encoding#==` and answer
|
|
287
|
+
/// `Encoding#ascii_compatible?` through Ruby method calls, which cost more
|
|
288
|
+
/// than most path operations; this compares indexes, and remembers whether
|
|
289
|
+
/// an encoding is ASCII-compatible (which never changes).
|
|
290
|
+
#[derive(Clone, Copy, PartialEq, Eq)]
|
|
291
|
+
pub struct EncodingId(EncodingIndex);
|
|
292
|
+
|
|
293
|
+
static UTF8_INDEX: AtomicI32 = AtomicI32::new(-1);
|
|
294
|
+
static US_ASCII_INDEX: AtomicI32 = AtomicI32::new(-1);
|
|
295
|
+
static ASCII_8BIT_INDEX: AtomicI32 = AtomicI32::new(-1);
|
|
296
|
+
|
|
297
|
+
const UNKNOWN: u8 = 0;
|
|
298
|
+
const ASCII_COMPATIBLE: u8 = 1;
|
|
299
|
+
const NOT_ASCII_COMPATIBLE: u8 = 2;
|
|
300
|
+
#[allow(clippy::declare_interior_mutable_const)]
|
|
301
|
+
const UNKNOWN_SLOT: AtomicU8 = AtomicU8::new(UNKNOWN);
|
|
302
|
+
static ASCII_COMPATIBILITY: [AtomicU8; 128] = [UNKNOWN_SLOT; 128];
|
|
303
|
+
|
|
304
|
+
impl EncodingId {
|
|
305
|
+
/// The encoding of a `String`.
|
|
306
|
+
pub fn of(string: &RString) -> Self {
|
|
307
|
+
// Safety: `string` is a `String`.
|
|
308
|
+
EncodingId(unsafe { rubysys::encoding::rb_enc_get_index(string.value()) })
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
pub fn from_encoding(encoding: &Encoding) -> Self {
|
|
312
|
+
EncodingId(encoding.index())
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
pub fn filesystem() -> Self {
|
|
316
|
+
EncodingId::from_encoding(&Encoding::filesystem())
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
pub fn utf8() -> Self {
|
|
320
|
+
EncodingId::known(&UTF8_INDEX, Encoding::utf8)
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
pub fn us_ascii() -> Self {
|
|
324
|
+
EncodingId::known(&US_ASCII_INDEX, Encoding::us_ascii)
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
pub fn ascii_8bit() -> Self {
|
|
328
|
+
EncodingId::known(&ASCII_8BIT_INDEX, Encoding::ascii_8bit)
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
fn known(cache: &AtomicI32, encoding: fn() -> Encoding) -> Self {
|
|
332
|
+
let index = cache.load(Ordering::Relaxed);
|
|
333
|
+
if index >= 0 {
|
|
334
|
+
return EncodingId(index);
|
|
335
|
+
}
|
|
336
|
+
let index = encoding().index();
|
|
337
|
+
cache.store(index, Ordering::Relaxed);
|
|
338
|
+
EncodingId(index)
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
/// The `Encoding` object.
|
|
342
|
+
pub fn to_encoding(self) -> Encoding {
|
|
343
|
+
// Safety: the index is valid (see `is_char_boundary`).
|
|
344
|
+
Encoding::from(unsafe {
|
|
345
|
+
rubysys::encoding::rb_enc_from_encoding(rubysys::encoding::rb_enc_from_index(self.0))
|
|
346
|
+
})
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
pub fn name(self) -> String {
|
|
350
|
+
self.to_encoding().name()
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
pub fn is_ascii_compatible(self) -> bool {
|
|
354
|
+
let slot = usize::try_from(self.0).ok().and_then(|index| ASCII_COMPATIBILITY.get(index));
|
|
355
|
+
match slot.map(|slot| slot.load(Ordering::Relaxed)) {
|
|
356
|
+
Some(ASCII_COMPATIBLE) => true,
|
|
357
|
+
Some(NOT_ASCII_COMPATIBLE) => false,
|
|
358
|
+
_ => {
|
|
359
|
+
let compatible = self.to_encoding().is_ascii_compatible();
|
|
360
|
+
if let Some(slot) = slot {
|
|
361
|
+
slot.store(if compatible { ASCII_COMPATIBLE } else { NOT_ASCII_COMPATIBLE }, Ordering::Relaxed);
|
|
362
|
+
}
|
|
363
|
+
compatible
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
// UTF-8, or an encoding of single bytes where every character starts a
|
|
369
|
+
// character.
|
|
370
|
+
fn is_single_byte_or_utf8(self) -> bool {
|
|
371
|
+
self == EncodingId::utf8() || self == EncodingId::us_ascii() || self == EncodingId::ascii_8bit()
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
// The length of the character at `start` (`rb_enc_mbclen`), at least 1.
|
|
375
|
+
fn char_len(self, bytes: &[u8], start: usize) -> usize {
|
|
376
|
+
// Safety: the index is valid (see `is_char_boundary`), and both
|
|
377
|
+
// pointers are within `bytes`, `end` one past its last byte.
|
|
378
|
+
let length = unsafe {
|
|
379
|
+
rubysys::encoding::rb_enc_mbclen(
|
|
380
|
+
bytes[start..].as_ptr() as *const c_char,
|
|
381
|
+
bytes.as_ptr_range().end as *const c_char,
|
|
382
|
+
rubysys::encoding::rb_enc_from_index(self.0),
|
|
383
|
+
)
|
|
384
|
+
};
|
|
385
|
+
(length.max(1) as usize).min(bytes.len() - start)
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/// In encodings such as Shift_JIS, `\` (0x5C) can be the second byte of a
|
|
389
|
+
/// character. Windows Ruby only takes a `\` that starts a character for a
|
|
390
|
+
/// separator; this returns a copy of `bytes` with the others replaced by
|
|
391
|
+
/// NUL (which a path never contains), or `None` if there are none.
|
|
392
|
+
pub fn mask_backslashes(self, bytes: &[u8]) -> Option<Vec<u8>> {
|
|
393
|
+
if self.is_single_byte_or_utf8() || !bytes.contains(&b'\\') {
|
|
394
|
+
return None;
|
|
395
|
+
}
|
|
396
|
+
let mut masked: Option<Vec<u8>> = None;
|
|
397
|
+
let mut index = 0;
|
|
398
|
+
while index < bytes.len() {
|
|
399
|
+
let length = self.char_len(bytes, index);
|
|
400
|
+
for trailing in index + 1..index + length {
|
|
401
|
+
if bytes[trailing] == b'\\' {
|
|
402
|
+
masked.get_or_insert_with(|| bytes.to_vec())[trailing] = 0;
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
index += length;
|
|
406
|
+
}
|
|
407
|
+
masked
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
/// `bytes`, in this encoding, converted to the encoding `to`
|
|
411
|
+
/// (`String#encode`), or `None` if they can't be.
|
|
412
|
+
#[cfg_attr(unix, allow(dead_code))]
|
|
413
|
+
pub fn transcode(self, bytes: &[u8], to: EncodingId) -> Option<Vec<u8>> {
|
|
414
|
+
let string = self.new_string(bytes);
|
|
415
|
+
let encoded = string.protect_send("encode", &[to.to_encoding().into()]).ok()?;
|
|
416
|
+
// Copied before anything else can run the garbage collector
|
|
417
|
+
let encoded = encoded.try_convert_to::<RString>().ok()?;
|
|
418
|
+
Some(encoded.to_bytes_unchecked().to_vec())
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
/// A new `String` of `bytes` in this encoding.
|
|
422
|
+
pub fn new_string(self, bytes: &[u8]) -> RString {
|
|
423
|
+
let bytes = unmask(bytes);
|
|
424
|
+
// Safety: `bytes` is valid for its length, which Ruby copies, and the
|
|
425
|
+
// index is valid (see `is_char_boundary`).
|
|
426
|
+
RString::from(unsafe {
|
|
427
|
+
rubysys::string::rb_enc_str_new(
|
|
428
|
+
bytes.as_ptr() as *const c_char,
|
|
429
|
+
bytes.len() as c_long,
|
|
430
|
+
rubysys::encoding::rb_enc_from_index(self.0),
|
|
431
|
+
)
|
|
432
|
+
})
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
/// The encoding of a string built from parts of other strings, and whether
|
|
437
|
+
/// all of those were ASCII only.
|
|
438
|
+
pub struct EncodingOf {
|
|
439
|
+
pub encoding: EncodingId,
|
|
440
|
+
ascii_only: bool,
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
impl EncodingOf {
|
|
444
|
+
pub fn new(path: &PathString) -> Self {
|
|
445
|
+
EncodingOf::of_bytes(path.bytes(), path.encoding)
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
pub fn of_bytes(bytes: &[u8], encoding: EncodingId) -> Self {
|
|
449
|
+
EncodingOf { encoding, ascii_only: bytes.is_ascii() }
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
pub fn merge(self, path: &PathString) -> Result<Self, AnyException> {
|
|
453
|
+
self.merge_with(EncodingOf::new(path))
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
/// Ruby's rules for the encoding of two concatenated strings.
|
|
457
|
+
pub fn merge_with(self, other: EncodingOf) -> Result<Self, AnyException> {
|
|
458
|
+
if self.encoding == other.encoding || other.ascii_only {
|
|
459
|
+
Ok(EncodingOf { ascii_only: self.ascii_only && other.ascii_only, ..self })
|
|
460
|
+
} else if self.ascii_only {
|
|
461
|
+
Ok(other)
|
|
462
|
+
} else {
|
|
463
|
+
Err(error(
|
|
464
|
+
&Class::encoding_compatibility_error(),
|
|
465
|
+
&format!("incompatible character encodings: {} and {}", self.encoding.name(), other.encoding.name()),
|
|
466
|
+
))
|
|
467
|
+
}
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
/// `object.inspect`
|
|
472
|
+
pub fn inspect<T: Object>(object: &T) -> Result<String, AnyException> {
|
|
473
|
+
let inspected = object.protect_send("inspect", &[])?;
|
|
474
|
+
Ok(match inspected.try_convert_to::<RString>() {
|
|
475
|
+
Ok(string) => String::from_utf8_lossy(string.to_bytes_unchecked()).into_owned(),
|
|
476
|
+
Err(_) => String::new(),
|
|
477
|
+
})
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
/// Converts a path-like object to a `String` the way `FasterPath.join` and
|
|
481
|
+
/// `FasterPath.relative_path_from` always have: `String`s as they are,
|
|
482
|
+
/// `Pathname`s by their path, then `to_path`, then `to_s`.
|
|
483
|
+
pub fn path_like_to_string(object: &AnyObject, pathname_class: &Class) -> Result<RString, AnyException> {
|
|
484
|
+
if object.ty() == ValueType::RString {
|
|
485
|
+
return object.try_convert_to::<RString>();
|
|
486
|
+
}
|
|
487
|
+
if object.is_kind_of(pathname_class) {
|
|
488
|
+
let path = object.protect_send("to_s", &[])?;
|
|
489
|
+
return expect_string(path, object, "to_s");
|
|
490
|
+
}
|
|
491
|
+
let responds_to_path = object.protect_send("respond_to?", &[Symbol::new("to_path").into()])?;
|
|
492
|
+
let method = if responds_to_path.value().is_true() { "to_path" } else { "to_s" };
|
|
493
|
+
let path = object.protect_send(method, &[])?;
|
|
494
|
+
expect_string(path, object, method)
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
fn expect_string(result: AnyObject, object: &AnyObject, method: &str) -> Result<RString, AnyException> {
|
|
498
|
+
result.try_convert_to::<RString>().map_err(|_| {
|
|
499
|
+
let class_name = object.protect_send("class", &[]).ok().
|
|
500
|
+
and_then(|class| inspect(&class).ok()).unwrap_or_default();
|
|
501
|
+
error(
|
|
502
|
+
&Class::type_error(),
|
|
503
|
+
&format!("can't convert {} to String ({}#{} gives {})", class_name, class_name, method, result_class(&result)),
|
|
504
|
+
)
|
|
505
|
+
})
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
fn result_class(result: &AnyObject) -> String {
|
|
509
|
+
result.protect_send("class", &[]).ok().and_then(|class| inspect(&class).ok()).unwrap_or_default()
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
pub fn pathname_class() -> Result<Class, AnyException> {
|
|
513
|
+
Class::from_path("Pathname")
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
extern "C" {
|
|
517
|
+
fn rb_obj_alloc(klass: Value) -> Value;
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
/// A `Pathname` of `path`, a new `String` without null bytes.
|
|
521
|
+
///
|
|
522
|
+
/// This sets `@path` like `Pathname#initialize` does instead of calling
|
|
523
|
+
/// `Pathname.new`, which is much faster and runs no Ruby code.
|
|
524
|
+
pub fn new_pathname(pathname_class: &Class, path: RString) -> RubyResult {
|
|
525
|
+
// Safety: `pathname_class` is a class (`Class::from_path` checks), and
|
|
526
|
+
// `Pathname` uses the default allocator.
|
|
527
|
+
let mut pathname = AnyObject::from(unsafe { rb_obj_alloc(pathname_class.value()) });
|
|
528
|
+
pathname.instance_variable_set("@path", path);
|
|
529
|
+
Ok(pathname)
|
|
530
|
+
}
|
metadata
CHANGED
|
@@ -1,41 +1,41 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: faster_path
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.4.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Daniel P. Clark
|
|
8
|
-
autorequire:
|
|
8
|
+
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date:
|
|
11
|
+
date: 2026-10-07 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: bundler
|
|
15
15
|
requirement: !ruby/object:Gem::Requirement
|
|
16
16
|
requirements:
|
|
17
|
-
- - "
|
|
17
|
+
- - ">="
|
|
18
18
|
- !ruby/object:Gem::Version
|
|
19
19
|
version: '1.12'
|
|
20
20
|
type: :runtime
|
|
21
21
|
prerelease: false
|
|
22
22
|
version_requirements: !ruby/object:Gem::Requirement
|
|
23
23
|
requirements:
|
|
24
|
-
- - "
|
|
24
|
+
- - ">="
|
|
25
25
|
- !ruby/object:Gem::Version
|
|
26
26
|
version: '1.12'
|
|
27
27
|
- !ruby/object:Gem::Dependency
|
|
28
28
|
name: rake
|
|
29
29
|
requirement: !ruby/object:Gem::Requirement
|
|
30
30
|
requirements:
|
|
31
|
-
- - "
|
|
31
|
+
- - ">="
|
|
32
32
|
- !ruby/object:Gem::Version
|
|
33
33
|
version: '12.3'
|
|
34
34
|
type: :runtime
|
|
35
35
|
prerelease: false
|
|
36
36
|
version_requirements: !ruby/object:Gem::Requirement
|
|
37
37
|
requirements:
|
|
38
|
-
- - "
|
|
38
|
+
- - ">="
|
|
39
39
|
- !ruby/object:Gem::Version
|
|
40
40
|
version: '12.3'
|
|
41
41
|
- !ruby/object:Gem::Dependency
|
|
@@ -72,14 +72,14 @@ dependencies:
|
|
|
72
72
|
requirements:
|
|
73
73
|
- - "~>"
|
|
74
74
|
- !ruby/object:Gem::Version
|
|
75
|
-
version: '5.
|
|
75
|
+
version: '5.11'
|
|
76
76
|
type: :development
|
|
77
77
|
prerelease: false
|
|
78
78
|
version_requirements: !ruby/object:Gem::Requirement
|
|
79
79
|
requirements:
|
|
80
80
|
- - "~>"
|
|
81
81
|
- !ruby/object:Gem::Version
|
|
82
|
-
version: '5.
|
|
82
|
+
version: '5.11'
|
|
83
83
|
- !ruby/object:Gem::Dependency
|
|
84
84
|
name: minitest-reporters
|
|
85
85
|
requirement: !ruby/object:Gem::Requirement
|
|
@@ -163,24 +163,21 @@ files:
|
|
|
163
163
|
- src/chop_basename.rs
|
|
164
164
|
- src/cleanpath_aggressive.rs
|
|
165
165
|
- src/cleanpath_conservative.rs
|
|
166
|
-
- src/debug.rs
|
|
167
166
|
- src/dirname.rs
|
|
168
167
|
- src/extname.rs
|
|
169
|
-
- src/helpers.rs
|
|
170
168
|
- src/lib.rs
|
|
171
|
-
- src/memrnchr.rs
|
|
172
169
|
- src/path_parsing.rs
|
|
173
170
|
- src/pathname.rs
|
|
174
|
-
- src/pathname_sys.rs
|
|
175
171
|
- src/plus.rs
|
|
176
172
|
- src/prepend_prefix.rs
|
|
177
173
|
- src/relative_path_from.rs
|
|
174
|
+
- src/ruby.rs
|
|
178
175
|
- src/rust_arch_bits.rs
|
|
179
176
|
homepage: https://github.com/danielpclark/faster_path
|
|
180
177
|
licenses:
|
|
181
178
|
- MIT OR Apache-2.0
|
|
182
179
|
metadata: {}
|
|
183
|
-
post_install_message:
|
|
180
|
+
post_install_message:
|
|
184
181
|
rdoc_options: []
|
|
185
182
|
require_paths:
|
|
186
183
|
- lib
|
|
@@ -188,16 +185,18 @@ required_ruby_version: !ruby/object:Gem::Requirement
|
|
|
188
185
|
requirements:
|
|
189
186
|
- - ">="
|
|
190
187
|
- !ruby/object:Gem::Version
|
|
191
|
-
version: '
|
|
188
|
+
version: '2.5'
|
|
189
|
+
- - "<"
|
|
190
|
+
- !ruby/object:Gem::Version
|
|
191
|
+
version: '3.0'
|
|
192
192
|
required_rubygems_version: !ruby/object:Gem::Requirement
|
|
193
193
|
requirements:
|
|
194
194
|
- - ">="
|
|
195
195
|
- !ruby/object:Gem::Version
|
|
196
196
|
version: '0'
|
|
197
197
|
requirements: []
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
signing_key:
|
|
198
|
+
rubygems_version: 3.1.6
|
|
199
|
+
signing_key:
|
|
201
200
|
specification_version: 4
|
|
202
201
|
summary: Reimplementation of Pathname for better performance
|
|
203
202
|
test_files: []
|
data/src/debug.rs
DELETED
|
@@ -1,29 +0,0 @@
|
|
|
1
|
-
use ruru::{
|
|
2
|
-
RString,
|
|
3
|
-
AnyObject,
|
|
4
|
-
Object,
|
|
5
|
-
};
|
|
6
|
-
|
|
7
|
-
#[derive(Debug)]
|
|
8
|
-
pub struct RubyDebugInfo {
|
|
9
|
-
object_id: String,
|
|
10
|
-
class: String,
|
|
11
|
-
inspect: String,
|
|
12
|
-
}
|
|
13
|
-
|
|
14
|
-
impl From<AnyObject> for RubyDebugInfo {
|
|
15
|
-
fn from(ao: AnyObject) -> Self {
|
|
16
|
-
let object_id = ao.send("object_id", None).send("to_s", None).
|
|
17
|
-
try_convert_to::<RString>().unwrap_or(RString::new("Failed to get object_id!")).to_string();
|
|
18
|
-
let class = ao.send("class", None).send("to_s", None).
|
|
19
|
-
try_convert_to::<RString>().unwrap_or(RString::new("Failed to get class!")).to_string();
|
|
20
|
-
let inspect = ao.send("inspect", None).
|
|
21
|
-
try_convert_to::<RString>().unwrap_or(RString::new("Failed to get inspect!")).to_string();
|
|
22
|
-
|
|
23
|
-
RubyDebugInfo {
|
|
24
|
-
object_id: object_id,
|
|
25
|
-
class: class,
|
|
26
|
-
inspect: inspect,
|
|
27
|
-
}
|
|
28
|
-
}
|
|
29
|
-
}
|