faster_path 0.3.10 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/src/lib.rs CHANGED
@@ -4,16 +4,10 @@
4
4
  // http://apache.org/licenses/LICENSE-2.0> or the MIT license <LICENSE-MIT or
5
5
  // http://opensource.org/licenses/MIT>, at your option. This file may not be
6
6
  // copied, modified, or distributed except according to those terms.
7
- #[macro_use]
8
- extern crate ruru;
7
+ #![forbid(unsafe_op_in_unsafe_fn)]
9
8
 
10
9
  #[macro_use]
11
- extern crate lazy_static;
12
-
13
- module!(FasterPath);
14
-
15
- mod debug;
16
- mod helpers;
10
+ mod ruby;
17
11
  mod pathname;
18
12
  mod basename;
19
13
  mod chop_basename;
@@ -21,149 +15,106 @@ mod cleanpath_aggressive;
21
15
  mod cleanpath_conservative;
22
16
  mod dirname;
23
17
  mod extname;
24
- mod pathname_sys;
25
18
  mod plus;
26
19
  mod prepend_prefix;
27
20
  pub mod rust_arch_bits;
28
- mod memrnchr;
29
21
  mod path_parsing;
30
22
  mod relative_path_from;
31
23
 
32
- use pathname::Pathname;
33
- use pathname_sys::raise;
34
-
35
- use ruru::{Module, Object, RString, Boolean, AnyObject};
36
-
37
- use pathname_sys::*;
38
-
39
- methods!(
40
- FasterPath,
41
- _itself,
24
+ use rutie::{Class, Module, Object, RString};
42
25
 
43
- fn pub_add_trailing_separator(pth: RString) -> RString {
44
- pathname::pn_add_trailing_separator(pth)
26
+ ruby_methods! {
27
+ fn pub_add_trailing_separator(arguments) {
28
+ pathname::pn_add_trailing_separator(arguments)
45
29
  }
46
30
 
47
- fn pub_is_absolute(pth: RString) -> Boolean {
48
- pathname::pn_is_absolute(pth)
31
+ fn pub_is_absolute(arguments) {
32
+ pathname::pn_is_absolute(arguments)
49
33
  }
50
34
 
51
- // fn r_ascend(){}
52
-
53
- fn pub_basename(pth: RString, ext: RString) -> RString {
54
- pathname::pn_basename(pth, ext)
35
+ fn pub_basename(arguments) {
36
+ pathname::pn_basename(arguments)
55
37
  }
56
38
 
57
- fn pub_children(pth: RString, with_dir: Boolean) -> AnyObject {
58
- pathname::pn_children(pth, with_dir).
59
- map_err(|e| raise(e) ).unwrap()
39
+ fn pub_children(arguments) {
40
+ pathname::pn_children(arguments)
60
41
  }
61
42
 
62
- fn pub_children_compat(pth: RString, with_dir: Boolean) -> AnyObject {
63
- pathname::pn_children_compat(pth, with_dir).
64
- map_err(|e| raise(e) ).unwrap()
43
+ fn pub_children_compat(arguments) {
44
+ pathname::pn_children_compat(arguments)
65
45
  }
66
46
 
67
- fn pub_chop_basename(pth: RString) -> AnyObject {
68
- pathname::pn_chop_basename(pth)
47
+ fn pub_chop_basename(arguments) {
48
+ pathname::pn_chop_basename(arguments)
69
49
  }
70
50
 
71
- // fn r_cleanpath(){ pub_cleanpath(r_to_path()) }
72
- // fn pub_cleanpath(pth: RString){}
73
-
74
- fn pub_cleanpath_aggressive(pth: RString) -> RString {
75
- pathname::pn_cleanpath_aggressive(pth)
51
+ fn pub_cleanpath_aggressive(arguments) {
52
+ pathname::pn_cleanpath_aggressive(arguments)
76
53
  }
77
54
 
78
- fn pub_cleanpath_conservative(pth: RString) -> RString {
79
- pathname::pn_cleanpath_conservative(pth)
55
+ fn pub_cleanpath_conservative(arguments) {
56
+ pathname::pn_cleanpath_conservative(arguments)
80
57
  }
81
58
 
82
- fn pub_del_trailing_separator(pth: RString) -> RString {
83
- pathname::pn_del_trailing_separator(pth)
59
+ fn pub_del_trailing_separator(arguments) {
60
+ pathname::pn_del_trailing_separator(arguments)
84
61
  }
85
62
 
86
- // fn r_descend(){}
87
-
88
- fn pub_is_directory(pth: RString) -> Boolean {
89
- pathname::pn_is_directory(pth)
63
+ fn pub_is_directory(arguments) {
64
+ pathname::pn_is_directory(arguments)
90
65
  }
91
66
 
92
- fn pub_dirname(pth: RString) -> RString {
93
- pathname::pn_dirname(pth)
67
+ fn pub_dirname(arguments) {
68
+ pathname::pn_dirname(arguments)
94
69
  }
95
70
 
96
- // fn r_each_child(){}
97
- // fn pub_each_child(){}
98
-
99
- // fn pub_each_filename(pth: RString) -> NilClass {
100
- // pathname::pn_each_filename(pth)
101
- // }
102
-
103
71
  // pub_entries returns an array of String objects
104
- fn pub_entries(pth: RString) -> AnyObject {
105
- pathname::pn_entries(pth).
106
- map_err(|e| raise(e) ).unwrap()
72
+ fn pub_entries(arguments) {
73
+ pathname::pn_entries(arguments)
107
74
  }
108
75
 
109
76
  // pub_entries_compat returns an array of Pathname objects
110
- fn pub_entries_compat(pth: RString) -> AnyObject {
111
- pathname::pn_entries_compat(pth).
112
- map_err(|e| raise(e) ).unwrap()
77
+ fn pub_entries_compat(arguments) {
78
+ pathname::pn_entries_compat(arguments)
113
79
  }
114
80
 
115
- fn pub_extname(pth: RString) -> RString {
116
- pathname::pn_extname(pth)
81
+ fn pub_extname(arguments) {
82
+ pathname::pn_extname(arguments)
117
83
  }
118
84
 
119
- // fn r_find(ignore_error: Boolean){}
120
- // fn pub_find(pth: RString ,ignore_error: Boolean){}
121
-
122
- fn pub_has_trailing_separator(pth: RString) -> Boolean {
123
- pathname::pn_has_trailing_separator(pth)
85
+ fn pub_has_trailing_separator(arguments) {
86
+ pathname::pn_has_trailing_separator(arguments)
124
87
  }
125
88
 
126
- // fn pub_mkpath(pth: RString) -> NilClass {
127
- // pathname::pn_mkpath(pth)
128
- // }
129
-
130
- // fn r_is_mountpoint(){ pub_is_mountpount(r_to_path()) }
131
- // fn pub_is_mountpoint(pth: RString){}
132
-
133
- // fn r_parent(){ pub_parent(r_to_path()) }
134
- // fn pub_parent(pth: RString){}
135
-
136
- // also need impl +
137
- fn pub_plus(pth1: RString, pth2: RString) -> RString {
138
- pathname::pn_plus(pth1, pth2)
89
+ fn pub_join(arguments) {
90
+ pathname::pn_join(arguments)
139
91
  }
140
92
 
141
- // fn r_prepend_prefix(prefix: RString, relpath: RString){}
142
-
143
- fn pub_is_relative(pth: RString) -> Boolean {
144
- pathname::pn_is_relative(pth)
93
+ fn pub_plus(arguments) {
94
+ pathname::pn_plus(arguments)
145
95
  }
146
96
 
147
- // fn r_root(){ pub_root(r_to_path()) }
148
- // fn pub_root(pth: RString){}
149
-
150
- // fn r_split_names(pth: RString){}
151
-
152
- fn pub_relative_path_from(itself: RString, base_directory: AnyObject) -> Pathname {
153
- let to_string = |i: AnyObject| { RString::from(i.send("to_s", None).value()) };
154
-
155
- pathname::pn_relative_path_from(itself, base_directory.map(to_string)).
156
- map_err(|e| raise(e) ).unwrap()
97
+ fn pub_is_relative(arguments) {
98
+ pathname::pn_is_relative(arguments)
157
99
  }
158
100
 
159
- // fn pub_rmtree(pth: RString) -> NilClass {
160
- // pathname::pn_rmtree(pth)
161
- // }
162
- );
101
+ fn pub_relative_path_from(arguments) {
102
+ pathname::pn_relative_path_from(arguments)
103
+ }
104
+ }
163
105
 
164
106
  #[allow(non_snake_case)]
165
107
  #[no_mangle]
166
108
  pub extern "C" fn Init_faster_pathname() {
109
+ // An extension of only a dot: "" before Ruby 2.7, "." since. ("a." would
110
+ // tell on Unix, but Windows doesn't count trailing dots.)
111
+ let trailing_dot_extname = Class::file().
112
+ protect_send("extname", &[RString::new_utf8("a./").into()]).
113
+ ok().
114
+ and_then(|extname| extname.try_convert_to::<RString>().ok()).
115
+ map_or(false, |extname| extname.to_bytes_unchecked() == b".");
116
+ extname::set_trailing_dot_is_extname(trailing_dot_extname);
117
+
167
118
  Module::from_existing("FasterPath").define(|itself| {
168
119
  itself.def_self("absolute?", pub_is_absolute);
169
120
  itself.def_self("add_trailing_separator", pub_add_trailing_separator);
@@ -177,8 +128,7 @@ pub extern "C" fn Init_faster_pathname() {
177
128
  itself.def_self("entries_compat", pub_entries_compat);
178
129
  itself.def_self("extname", pub_extname);
179
130
  itself.def_self("has_trailing_separator?", pub_has_trailing_separator);
180
- //itself.def_self("join", pub_join);
181
- pathname_sys::define_singleton_method(itself.value(), "join", pub_join);
131
+ itself.def_self("join", pub_join);
182
132
  itself.def_self("plus", pub_plus);
183
133
  itself.def_self("relative?", pub_is_relative);
184
134
  itself.def_self("relative_path_from", pub_relative_path_from);
data/src/path_parsing.rs CHANGED
@@ -1,29 +1,290 @@
1
- extern crate memchr;
1
+ // The building blocks of Ruby's path handling (`file.c`), for both of the
2
+ // sets of rules Ruby is built with:
3
+ //
4
+ // * Unix: `/` is the only separator.
5
+ // * Windows ("DOSISH"): `/` and `\` are separators, a path can start with a
6
+ // drive letter (`C:`) or a UNC prefix (`//server/share`), NTFS ignores
7
+ // trailing dots and spaces and `:stream` names at the end of a file name,
8
+ // and file names compare case-insensitively.
9
+ //
10
+ // Both are always compiled so each can be tested on any platform;
11
+ // `Rules::NATIVE` is the one Ruby uses where this is built.
12
+ use memchr::{memchr, memchr2, memrchr, memrchr2};
2
13
 
3
- use self::memchr::{memchr, memrchr};
4
- use memrnchr::memrnchr;
5
- use std::path::MAIN_SEPARATOR;
6
- use std::str;
7
-
8
- pub const SEP: u8 = MAIN_SEPARATOR as u8;
9
- lazy_static! {
10
- pub static ref SEP_STR: &'static str = str::from_utf8(&[SEP]).unwrap();
14
+ #[derive(Clone, Copy, Debug, PartialEq, Eq)]
15
+ pub struct Rules {
16
+ dosish: bool,
11
17
  }
12
18
 
13
- // Returns the byte offset of the last byte that equals MAIN_SEPARATOR.
14
- #[inline(always)]
15
- pub fn find_last_sep_pos(bytes: &[u8]) -> Option<usize> {
16
- memrchr(SEP, bytes)
19
+ // `File::SEPARATOR`, which joins paths on every platform.
20
+ pub const SEP_BYTES: &[u8] = b"/";
21
+
22
+ impl Rules {
23
+ #[cfg_attr(not(test), allow(dead_code))]
24
+ pub const UNIX: Rules = Rules { dosish: false };
25
+ #[cfg_attr(not(test), allow(dead_code))]
26
+ pub const WINDOWS: Rules = Rules { dosish: true };
27
+ pub const NATIVE: Rules = Rules { dosish: cfg!(windows) };
28
+
29
+ #[inline(always)]
30
+ pub fn is_dosish(self) -> bool {
31
+ self.dosish
32
+ }
33
+
34
+ // Ruby's `isdirsep`
35
+ #[inline(always)]
36
+ pub fn is_sep(self, c: u8) -> bool {
37
+ c == b'/' || (self.dosish && c == b'\\')
38
+ }
39
+
40
+ // Byte offset of the last separator.
41
+ #[inline]
42
+ pub fn last_sep_pos(self, bytes: &[u8]) -> Option<usize> {
43
+ if self.dosish { memrchr2(b'/', b'\\', bytes) } else { memrchr(b'/', bytes) }
44
+ }
45
+
46
+ // Byte offset of the last byte that isn't a separator.
47
+ #[inline]
48
+ pub fn last_non_sep_pos(self, bytes: &[u8]) -> Option<usize> {
49
+ bytes.iter().rposition(|&c| !self.is_sep(c))
50
+ }
51
+
52
+ #[inline]
53
+ pub fn contains_sep(self, bytes: &[u8]) -> bool {
54
+ if self.dosish { memchr2(b'/', b'\\', bytes).is_some() } else { memchr(b'/', bytes).is_some() }
55
+ }
56
+
57
+ // `has_drive_letter`: "C:"
58
+ #[inline]
59
+ pub fn has_drive_letter(self, path: &[u8]) -> bool {
60
+ self.dosish && path.len() >= 2 && path[0].is_ascii_alphabetic() && path[1] == b':'
61
+ }
62
+
63
+ // `istrailinggarbage`: ignored at the end of NTFS file names.
64
+ #[inline(always)]
65
+ fn is_trailing_garbage(self, c: u8) -> bool {
66
+ self.dosish && (c == b'.' || c == b' ')
67
+ }
68
+
69
+ // `isADS`: starts an NTFS alternate data stream name.
70
+ #[inline(always)]
71
+ pub fn is_ads(self, c: u8) -> bool {
72
+ self.dosish && c == b':'
73
+ }
74
+
75
+ // `Pathname::SAME_PATHS`, and the comparison `File.basename` uses for
76
+ // extensions (`CASEFOLD_FILESYSTEM`).
77
+ #[inline]
78
+ pub fn same_path(self, a: &[u8], b: &[u8]) -> bool {
79
+ if self.dosish { a.eq_ignore_ascii_case(b) } else { a == b }
80
+ }
81
+
82
+ // `skipprefix`: the end of a UNC ("//server/share") or drive letter prefix.
83
+ pub fn skip_prefix(self, path: &[u8]) -> usize {
84
+ if !self.dosish {
85
+ return 0;
86
+ }
87
+ let end = path.len();
88
+ if end >= 2 && self.is_sep(path[0]) && self.is_sep(path[1]) {
89
+ let mut i = 2;
90
+ while i < end && self.is_sep(path[i]) {
91
+ i += 1;
92
+ }
93
+ i = self.next_sep(path, i);
94
+ if i + 1 < end && !self.is_sep(path[i + 1]) {
95
+ i = self.next_sep(path, i + 1);
96
+ }
97
+ return i;
98
+ }
99
+ if self.has_drive_letter(path) {
100
+ return 2;
101
+ }
102
+ 0
103
+ }
104
+
105
+ // `skiproot`: past a drive letter and the separators after it.
106
+ pub fn skip_root(self, path: &[u8]) -> usize {
107
+ let mut i = if self.has_drive_letter(path) { 2 } else { 0 };
108
+ while i < path.len() && self.is_sep(path[i]) {
109
+ i += 1;
110
+ }
111
+ i
112
+ }
113
+
114
+ // `nextdirsep`: the next separator at or after `from`, or the end.
115
+ fn next_sep(self, path: &[u8], from: usize) -> usize {
116
+ let rest = &path[from.min(path.len())..];
117
+ let found = if self.dosish { memchr2(b'/', b'\\', rest) } else { memchr(b'/', rest) };
118
+ found.map_or(path.len(), |pos| from + pos)
119
+ }
120
+
121
+ // `strrdirsep`: where the last run of separators that isn't at the end
122
+ // of the path starts.
123
+ pub fn last_separator(self, path: &[u8]) -> Option<usize> {
124
+ let end = self.last_non_sep_pos(path)? + 1;
125
+ let pos = self.last_sep_pos(&path[..end])?;
126
+ Some(path[..pos].iter().rposition(|&c| !self.is_sep(c)).map_or(0, |p| p + 1))
127
+ }
128
+
129
+ // `chompdirsep`: where the separators at the end of the path start, or
130
+ // its end.
131
+ #[inline]
132
+ pub fn chomp_dir_sep(self, path: &[u8]) -> usize {
133
+ self.last_non_sep_pos(path).map_or(0, |pos| pos + 1)
134
+ }
135
+
136
+ // `ntfs_tail`: the end of an NTFS file name without trailing dots,
137
+ // spaces, separators or `:stream`.
138
+ pub fn ntfs_tail(self, path: &[u8]) -> usize {
139
+ let end = path.len();
140
+ let mut i = 0;
141
+ while i < end && path[i] == b'.' {
142
+ i += 1;
143
+ }
144
+ while i < end && !self.is_ads(path[i]) {
145
+ if self.is_trailing_garbage(path[i]) {
146
+ let last = i;
147
+ i += 1;
148
+ while i < end && self.is_trailing_garbage(path[i]) {
149
+ i += 1;
150
+ }
151
+ if i >= end || self.is_ads(path[i]) {
152
+ return last;
153
+ }
154
+ } else if self.is_sep(path[i]) {
155
+ let last = i;
156
+ i += 1;
157
+ while i < end && self.is_sep(path[i]) {
158
+ i += 1;
159
+ }
160
+ if i >= end {
161
+ return last;
162
+ }
163
+ if self.is_ads(path[i]) {
164
+ i += 1;
165
+ }
166
+ } else {
167
+ i += 1;
168
+ }
169
+ }
170
+ i
171
+ }
172
+
173
+ // Ruby's Windows `stat` doesn't find a path whose first "..." is a
174
+ // whole name of dots, as Windows would take "dir\\..." for "dir"
175
+ // (`check_valid_dir` in `win32.c`).
176
+ pub fn has_dots_name(self, path: &[u8]) -> bool {
177
+ if !self.dosish {
178
+ return false;
179
+ }
180
+ let start = match memchr::memmem::find(path, b"...") {
181
+ Some(start) => start,
182
+ None => return false,
183
+ };
184
+ let end = start + path[start..].iter().take_while(|&&c| c == b'.').count();
185
+ let delimits = |c: u8| c == b':' || self.is_sep(c);
186
+ (start == 0 || delimits(path[start - 1])) && (end == path.len() || delimits(path[end]))
187
+ }
188
+
189
+ // `File.join(a, b)`
190
+ pub fn join(self, a: &[u8], b: &[u8]) -> Vec<u8> {
191
+ let tail = self.chomp_dir_sep(a);
192
+ if b.first().map_or(false, |&c| self.is_sep(c)) {
193
+ [&a[..tail], b].concat()
194
+ } else if tail == a.len() {
195
+ [a, SEP_BYTES, b].concat()
196
+ } else {
197
+ [a, b].concat()
198
+ }
199
+ }
17
200
  }
18
201
 
19
- // Returns the byte offset of the last byte that is not MAIN_SEPARATOR.
20
- #[inline(always)]
21
- pub fn find_last_non_sep_pos(bytes: &[u8]) -> Option<usize> {
22
- memrnchr(SEP, bytes)
202
+ // Where `part`, a slice of `whole`, starts in it.
203
+ #[inline]
204
+ pub fn offset_in(whole: &[u8], part: &[u8]) -> usize {
205
+ part.as_ptr() as usize - whole.as_ptr() as usize
23
206
  }
24
207
 
25
- // Whether the given byte sequence contains a MAIN_SEPARATOR.
26
- #[inline(always)]
27
- pub fn contains_sep(bytes: &[u8]) -> bool {
28
- memchr(SEP, bytes) != None
208
+ #[cfg(test)]
209
+ mod tests {
210
+ use super::*;
211
+
212
+ const U: Rules = Rules::UNIX;
213
+ const W: Rules = Rules::WINDOWS;
214
+
215
+ #[test]
216
+ fn it_finds_separators() {
217
+ assert_eq!(U.last_sep_pos(b""), None);
218
+ assert_eq!(U.last_sep_pos(b"a/b/c"), Some(3));
219
+ assert_eq!(U.last_sep_pos(b"a/b\\c"), Some(1));
220
+ assert_eq!(W.last_sep_pos(b"a/b\\c"), Some(3));
221
+ assert_eq!(U.last_non_sep_pos(b"///"), None);
222
+ assert_eq!(U.last_non_sep_pos(b"a///"), Some(0));
223
+ assert_eq!(W.last_non_sep_pos(b"a/\\/"), Some(0));
224
+ assert!(U.contains_sep(b"a/b"));
225
+ assert!(!U.contains_sep(b"a\\b"));
226
+ assert!(W.contains_sep(b"a\\b"));
227
+ }
228
+
229
+ #[test]
230
+ fn it_skips_prefixes() {
231
+ assert_eq!(U.skip_prefix(b"//a/b/c"), 0);
232
+ assert_eq!(W.skip_prefix(b"//a/b/c"), 5);
233
+ assert_eq!(W.skip_prefix(b"\\\\a\\b\\c"), 5);
234
+ assert_eq!(W.skip_prefix(b"//a/"), 3);
235
+ assert_eq!(W.skip_prefix(b"//a"), 3);
236
+ assert_eq!(W.skip_prefix(b"//"), 2);
237
+ assert_eq!(W.skip_prefix(b"C:/a"), 2);
238
+ assert_eq!(W.skip_prefix(b"/a"), 0);
239
+ assert_eq!(W.skip_root(b"C://a"), 4);
240
+ assert_eq!(U.skip_root(b"C://a"), 0);
241
+ assert_eq!(U.skip_root(b"//a"), 2);
242
+ }
243
+
244
+ #[test]
245
+ fn it_finds_the_last_separator() {
246
+ assert_eq!(U.last_separator(b"a//b//"), Some(1));
247
+ assert_eq!(U.last_separator(b"//b"), Some(0));
248
+ assert_eq!(U.last_separator(b"b//"), None);
249
+ assert_eq!(W.last_separator(b"a\\/b"), Some(1));
250
+ assert_eq!(U.chomp_dir_sep(b"a//"), 1);
251
+ assert_eq!(U.chomp_dir_sep(b"///"), 0);
252
+ assert_eq!(U.chomp_dir_sep(b"a"), 1);
253
+ }
254
+
255
+ #[test]
256
+ fn it_finds_ntfs_tails() {
257
+ assert_eq!(W.ntfs_tail(b"foo."), 3);
258
+ assert_eq!(W.ntfs_tail(b"foo. ."), 3);
259
+ assert_eq!(W.ntfs_tail(b"foo::$DATA"), 3);
260
+ assert_eq!(W.ntfs_tail(b"foo.bar"), 7);
261
+ assert_eq!(W.ntfs_tail(b"..."), 3);
262
+ assert_eq!(W.ntfs_tail(b"foo/"), 3);
263
+ }
264
+
265
+ #[test]
266
+ fn it_finds_names_of_dots() {
267
+ assert!(W.has_dots_name(b"dir/..."));
268
+ assert!(W.has_dots_name(b"..."));
269
+ assert!(W.has_dots_name(b"C:..../a"));
270
+ assert!(W.has_dots_name(b"a\\...\\b"));
271
+ assert!(!W.has_dots_name(b"dir/a..."));
272
+ assert!(!W.has_dots_name(b"dir/..a"));
273
+ assert!(!W.has_dots_name(b"dir/."));
274
+ assert!(!U.has_dots_name(b"dir/..."));
275
+ }
276
+
277
+ #[test]
278
+ fn it_joins_like_file_join() {
279
+ assert_eq!(U.join(b"a", b"b"), b"a/b");
280
+ assert_eq!(U.join(b"a/", b"b"), b"a/b");
281
+ assert_eq!(U.join(b"a//", b"b"), b"a//b");
282
+ assert_eq!(U.join(b"a/", b"/b"), b"a/b");
283
+ assert_eq!(U.join(b"a", b""), b"a/");
284
+ assert_eq!(U.join(b"/", b""), b"/");
285
+ assert_eq!(U.join(b"", b"b"), b"/b");
286
+ assert_eq!(W.join(b"a\\", b"b"), b"a\\b");
287
+ assert_eq!(W.join(b"a", b"\\b"), b"a\\b");
288
+ assert_eq!(U.join(b"a", b"\\b"), b"a/\\b");
289
+ }
29
290
  }