corvus_json_schema 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/Cargo.lock +390 -0
- data/Cargo.toml +8 -0
- data/LICENSE +201 -0
- data/README.md +111 -0
- data/VERSIONHISTORY.md +7 -0
- data/ext/corvus_json_schema/Cargo.toml +18 -0
- data/ext/corvus_json_schema/extconf.rb +6 -0
- data/ext/corvus_json_schema/rustfmt.toml +2 -0
- data/ext/corvus_json_schema/src/lib.rs +643 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/Cargo.toml +35 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/LICENSE +201 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/README.md +139 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/compiler.rs +1118 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/dialect.rs +108 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/document.rs +905 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan/fused.rs +1187 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan.rs +1728 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval.rs +1451 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/formats.rs +1021 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/instance.rs +228 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/lib.rs +189 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/loader.rs +403 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/metaschemas.rs +29 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/node.rs +417 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/numbers.rs +244 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/options.rs +108 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/pattern.rs +1591 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/results.rs +316 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/uri.rs +274 -0
- data/lib/corvus_json_schema/version.rb +5 -0
- data/lib/corvus_json_schema.rb +65 -0
- metadata +94 -0
|
@@ -0,0 +1,1591 @@
|
|
|
1
|
+
//! `pattern`/`patternProperties` matching with ECMA-262 semantics (the `u` flag), as JSON Schema specifies.
|
|
2
|
+
//!
|
|
3
|
+
//! A pattern gets the cheapest matcher that decides it exactly:
|
|
4
|
+
//!
|
|
5
|
+
//! - patterns every string matches (`""`, `.*`, `[\s\S]*` and the like), `.+`, and line lengths (`^.{1,256}$`);
|
|
6
|
+
//! - anchored sequences of quantified ASCII character classes and literals (`^[a-z][a-z0-9_]{0,29}$`, `^x-`,
|
|
7
|
+
//! `^[@$_#]`), matched in one pass over the string (the TypeScript evaluator's class-sequence fast path);
|
|
8
|
+
//! - alternatives of such sequences once groups are multiplied out (`^([a|A]uto)|([n|N]one)$`,
|
|
9
|
+
//! `^[Ee][Ss]2015(\.([Cc]ore|[Pp]roxy))?$`), sets of literals, and separated lists (`^([a-z]+)(\.[a-z]+)*$`),
|
|
10
|
+
//! as the C# evaluator's pattern matchers do;
|
|
11
|
+
//! - patterns within the common subset of ECMA-262 and the `regex` crate's syntax, translated with ECMA semantics
|
|
12
|
+
//! (`\d`/`\w` ASCII-only, ECMA's `\s` and `.`) and run by `regex`, which searches in linear time without
|
|
13
|
+
//! allocating per match;
|
|
14
|
+
//! - anything else (lookarounds, backreferences, word boundaries, Unicode properties) by `regress`, a backtracking
|
|
15
|
+
//! ECMA-262 engine.
|
|
16
|
+
//!
|
|
17
|
+
//! Patterns compile once per process (a pattern is immutable, so identical patterns share one matcher).
|
|
18
|
+
|
|
19
|
+
use std::collections::HashMap;
|
|
20
|
+
use std::fmt;
|
|
21
|
+
use std::sync::{Arc, LazyLock, Mutex};
|
|
22
|
+
|
|
23
|
+
/// A compiled `pattern`.
|
|
24
|
+
pub(crate) struct Pattern {
|
|
25
|
+
pub source: String,
|
|
26
|
+
matcher: Matcher,
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
enum Matcher {
|
|
30
|
+
/// Matches every string.
|
|
31
|
+
Everything,
|
|
32
|
+
/// `^literal` (with `$`: the whole string).
|
|
33
|
+
Literal {
|
|
34
|
+
text: Box<str>,
|
|
35
|
+
whole: bool,
|
|
36
|
+
},
|
|
37
|
+
Sequence(Sequence),
|
|
38
|
+
SeparatedList(SeparatedList),
|
|
39
|
+
/// `.+` (`^.+` with `start`): some (the first) character is not a line terminator.
|
|
40
|
+
HasContent {
|
|
41
|
+
start: bool,
|
|
42
|
+
},
|
|
43
|
+
/// `^.{min,max}$`: between `min` and `max` characters, none a line terminator.
|
|
44
|
+
Line {
|
|
45
|
+
min: u32,
|
|
46
|
+
max: u32,
|
|
47
|
+
},
|
|
48
|
+
/// `^(a|b|...)$`: one of a set of strings.
|
|
49
|
+
Literals(Literals),
|
|
50
|
+
/// Top-level alternatives of literals, each optionally anchored (`^a|b|c$`).
|
|
51
|
+
Alternatives(Box<[Alternative]>),
|
|
52
|
+
/// `^(?=[^SET]+$)(?=(.*\w)).+$`: a non-empty line without a character of the set, containing a word character.
|
|
53
|
+
/// With `bangs` (`^(?=!+[^SET]+$)…`, the set holding `!`), that line follows one or more `!`.
|
|
54
|
+
ExcludedClassWithWord {
|
|
55
|
+
set: CharSet,
|
|
56
|
+
bangs: bool,
|
|
57
|
+
},
|
|
58
|
+
Regex(regex::Regex),
|
|
59
|
+
/// A pattern that is valid only without the `u` flag (an identity escape such as `\&`), so matching UTF-16 code
|
|
60
|
+
/// units: translated for `regex`, which decides it exactly on strings with no character beyond the Basic
|
|
61
|
+
/// Multilingual Plane (where code units are characters); regress decides the rest.
|
|
62
|
+
RegexBmp(regex::Regex, regress::Regex),
|
|
63
|
+
Regress(regress::Regex),
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
impl fmt::Debug for Pattern {
|
|
67
|
+
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
|
68
|
+
write!(f, "Pattern({:?})", self.source)
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
impl Pattern {
|
|
73
|
+
#[inline]
|
|
74
|
+
pub fn is_match(&self, s: &str) -> bool {
|
|
75
|
+
match &self.matcher {
|
|
76
|
+
Matcher::Everything => true,
|
|
77
|
+
Matcher::Literal { text, whole } => {
|
|
78
|
+
if *whole {
|
|
79
|
+
s == &**text
|
|
80
|
+
} else {
|
|
81
|
+
s.starts_with(&**text)
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
Matcher::Sequence(seq) => seq.is_match(s),
|
|
85
|
+
Matcher::SeparatedList(list) => list.is_match(s),
|
|
86
|
+
Matcher::HasContent { start } => {
|
|
87
|
+
if *start {
|
|
88
|
+
s.chars().next().is_some_and(|c| !is_line_terminator(c))
|
|
89
|
+
} else {
|
|
90
|
+
s.chars().any(|c| !is_line_terminator(c))
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
Matcher::Line { min, max } => line_length(s).is_some_and(|n| n >= *min as usize && n <= *max as usize),
|
|
94
|
+
Matcher::Literals(set) => set.contains(s),
|
|
95
|
+
Matcher::Alternatives(alts) => {
|
|
96
|
+
let ascii = s.is_ascii();
|
|
97
|
+
alts.iter().any(|a| a.is_match(s, ascii))
|
|
98
|
+
}
|
|
99
|
+
Matcher::ExcludedClassWithWord { set, bangs } => {
|
|
100
|
+
let mut word = false;
|
|
101
|
+
let s = if *bangs {
|
|
102
|
+
let rest = s.trim_start_matches('!');
|
|
103
|
+
if rest.len() == s.len() || rest.is_empty() {
|
|
104
|
+
return false;
|
|
105
|
+
}
|
|
106
|
+
rest
|
|
107
|
+
} else {
|
|
108
|
+
s
|
|
109
|
+
};
|
|
110
|
+
for c in s.chars() {
|
|
111
|
+
if set.contains(c) || is_line_terminator(c) {
|
|
112
|
+
return false;
|
|
113
|
+
}
|
|
114
|
+
word |= CharSet::WORD.contains(c);
|
|
115
|
+
}
|
|
116
|
+
word
|
|
117
|
+
}
|
|
118
|
+
Matcher::Regex(re) => re.is_match(s),
|
|
119
|
+
// A UTF-8 lead byte of 0xF0 or above starts a character beyond the BMP.
|
|
120
|
+
Matcher::RegexBmp(re, fallback) => {
|
|
121
|
+
if s.bytes().any(|b| b >= 0xF0) {
|
|
122
|
+
fallback.find(s).is_some()
|
|
123
|
+
} else {
|
|
124
|
+
re.is_match(s)
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
Matcher::Regress(re) => re.find(s).is_some(),
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/// ECMA-262 `LineTerminator`, which `.` does not match.
|
|
133
|
+
#[inline]
|
|
134
|
+
fn is_line_terminator(c: char) -> bool {
|
|
135
|
+
matches!(c, '\n' | '\r' | '\u{2028}' | '\u{2029}')
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/// The number of characters, when none is a line terminator.
|
|
139
|
+
#[inline]
|
|
140
|
+
fn line_length(s: &str) -> Option<usize> {
|
|
141
|
+
if s.is_ascii() {
|
|
142
|
+
return (!s.bytes().any(|b| b == b'\n' || b == b'\r')).then_some(s.len());
|
|
143
|
+
}
|
|
144
|
+
let mut n = 0;
|
|
145
|
+
for c in s.chars() {
|
|
146
|
+
if is_line_terminator(c) {
|
|
147
|
+
return None;
|
|
148
|
+
}
|
|
149
|
+
n += 1;
|
|
150
|
+
}
|
|
151
|
+
Some(n)
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/// A set of strings: compared in turn when there are few, hashed otherwise.
|
|
155
|
+
enum Literals {
|
|
156
|
+
Few(Box<[Box<str>]>),
|
|
157
|
+
Many(std::collections::HashSet<Box<str>>),
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
impl Literals {
|
|
161
|
+
fn new(texts: Vec<String>) -> Literals {
|
|
162
|
+
if texts.len() <= 8 {
|
|
163
|
+
Literals::Few(texts.into_iter().map(String::into_boxed_str).collect())
|
|
164
|
+
} else {
|
|
165
|
+
Literals::Many(texts.into_iter().map(String::into_boxed_str).collect())
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
#[inline]
|
|
170
|
+
fn contains(&self, s: &str) -> bool {
|
|
171
|
+
match self {
|
|
172
|
+
Literals::Few(texts) => texts.iter().any(|t| **t == *s),
|
|
173
|
+
Literals::Many(set) => set.contains(s),
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/// One alternative of [`Matcher::Alternatives`]: a literal anchored at either end (or neither), a class sequence
|
|
179
|
+
/// anchored at the start, or a fixed-width class sequence anchored at the end.
|
|
180
|
+
enum Alternative {
|
|
181
|
+
Literal {
|
|
182
|
+
text: Box<str>,
|
|
183
|
+
start: bool,
|
|
184
|
+
end: bool,
|
|
185
|
+
},
|
|
186
|
+
/// `^sequence` (with `$` when the sequence's `to_end`).
|
|
187
|
+
Start(Sequence),
|
|
188
|
+
/// `sequence$` of `width` characters: matched over the string's last `width` characters.
|
|
189
|
+
End {
|
|
190
|
+
seq: Sequence,
|
|
191
|
+
width: usize,
|
|
192
|
+
},
|
|
193
|
+
/// An unanchored `sequence` of `width` characters: matched at each position.
|
|
194
|
+
Anywhere {
|
|
195
|
+
seq: Sequence,
|
|
196
|
+
width: usize,
|
|
197
|
+
},
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
impl Alternative {
|
|
201
|
+
/// Whether the alternative matches `s`, `ascii` saying whether `s` is ASCII.
|
|
202
|
+
#[inline]
|
|
203
|
+
fn is_match(&self, s: &str, ascii: bool) -> bool {
|
|
204
|
+
match self {
|
|
205
|
+
Alternative::Literal { text, start, end } => match (start, end) {
|
|
206
|
+
(true, true) => s == &**text,
|
|
207
|
+
(true, false) => s.starts_with(&**text),
|
|
208
|
+
(false, true) => s.ends_with(&**text),
|
|
209
|
+
(false, false) => s.contains(&**text),
|
|
210
|
+
},
|
|
211
|
+
Alternative::Start(seq) => {
|
|
212
|
+
if ascii {
|
|
213
|
+
seq.is_match_ascii(s.as_bytes())
|
|
214
|
+
} else {
|
|
215
|
+
seq.is_match_chars(s)
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
Alternative::End { seq, width } => {
|
|
219
|
+
if ascii {
|
|
220
|
+
return s.len().checked_sub(*width).is_some_and(|at| seq.is_match_ascii(&s.as_bytes()[at..]));
|
|
221
|
+
}
|
|
222
|
+
let at = {
|
|
223
|
+
let n = s.chars().count();
|
|
224
|
+
n.checked_sub(*width).map(|skip| s.char_indices().nth(skip).map_or(s.len(), |(i, _)| i))
|
|
225
|
+
};
|
|
226
|
+
at.is_some_and(|at| seq.is_match_chars(&s[at..]))
|
|
227
|
+
}
|
|
228
|
+
Alternative::Anywhere { seq, width } => {
|
|
229
|
+
if ascii {
|
|
230
|
+
let b = s.as_bytes();
|
|
231
|
+
return b.len() >= *width && (0..=b.len() - width).any(|at| seq.is_match_ascii(&b[at..]));
|
|
232
|
+
}
|
|
233
|
+
let n = s.chars().count();
|
|
234
|
+
n >= *width && s.char_indices().take(n - width + 1).any(|(at, _)| seq.is_match_chars(&s[at..]))
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/// Compiles a pattern with the `u` flag, or (as many validators accept them) without it; the flag is returned.
|
|
241
|
+
fn compile_regress(pattern: &str) -> Option<(regress::Regex, bool)> {
|
|
242
|
+
match regress::Regex::with_flags(pattern, "u") {
|
|
243
|
+
Ok(re) => Some((re, true)),
|
|
244
|
+
Err(_) => regress::Regex::new(pattern).ok().map(|re| (re, false)),
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
static CACHE: LazyLock<Mutex<HashMap<String, Arc<Pattern>>>> = LazyLock::new(|| Mutex::new(HashMap::new()));
|
|
249
|
+
|
|
250
|
+
/// Compiles (or fetches from the process-wide cache) a pattern; `None` when it is not a valid ECMA-262 regex.
|
|
251
|
+
pub(crate) fn compile(pattern: &str) -> Option<Arc<Pattern>> {
|
|
252
|
+
if let Some(p) = CACHE.lock().unwrap().get(pattern) {
|
|
253
|
+
return Some(p.clone());
|
|
254
|
+
}
|
|
255
|
+
// Validity is ECMA-262's: a pattern regress rejects is an error, whichever matcher would run it.
|
|
256
|
+
let (regress, unicode) = compile_regress(pattern)?;
|
|
257
|
+
let matcher = match choose(pattern, unicode) {
|
|
258
|
+
Some(m) => m,
|
|
259
|
+
None if !unicode => match translate(pattern, false).and_then(|t| regex::Regex::new(&t).ok()) {
|
|
260
|
+
Some(re) => Matcher::RegexBmp(re, regress),
|
|
261
|
+
None => Matcher::Regress(regress),
|
|
262
|
+
},
|
|
263
|
+
None => Matcher::Regress(regress),
|
|
264
|
+
};
|
|
265
|
+
let p = Arc::new(Pattern { source: pattern.to_string(), matcher });
|
|
266
|
+
CACHE.lock().unwrap().insert(pattern.to_string(), p.clone());
|
|
267
|
+
Some(p)
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
fn choose(pattern: &str, unicode: bool) -> Option<Matcher> {
|
|
271
|
+
// Unanchored (or start-anchored) `.*` finds an empty match in any string; `^.*$` does not (`.` stops at a line
|
|
272
|
+
// terminator), so it is not listed.
|
|
273
|
+
if matches!(pattern, "" | ".*" | "^.*" | ".*$" | "(.*)" | "^(.*)" | "[\\s\\S]*" | "^[\\s\\S]*" | "^[\\s\\S]*$") {
|
|
274
|
+
return Some(Matcher::Everything);
|
|
275
|
+
}
|
|
276
|
+
// Without the `u` flag, a pattern matches UTF-16 code units rather than characters: only regress has that.
|
|
277
|
+
if !unicode {
|
|
278
|
+
return None;
|
|
279
|
+
}
|
|
280
|
+
match pattern {
|
|
281
|
+
".+" | "." | "(.+)" => return Some(Matcher::HasContent { start: false }),
|
|
282
|
+
"^.+" | "^." => return Some(Matcher::HasContent { start: true }),
|
|
283
|
+
_ => {}
|
|
284
|
+
}
|
|
285
|
+
// `^X.*` (no `$`) matches exactly where `^X` does: `.*` can match nothing.
|
|
286
|
+
if let Some(rest) = pattern.strip_suffix(".*")
|
|
287
|
+
&& rest.starts_with('^')
|
|
288
|
+
&& rest.len() > 1
|
|
289
|
+
&& !ends_with_escape(rest)
|
|
290
|
+
&& !rest.ends_with(['*', '+', '?', '}', '|', '(', '^'])
|
|
291
|
+
&& let Some(m) = choose(rest, unicode)
|
|
292
|
+
{
|
|
293
|
+
return Some(m);
|
|
294
|
+
}
|
|
295
|
+
if let Some((min, max)) = line_range(pattern) {
|
|
296
|
+
return Some(Matcher::Line { min, max });
|
|
297
|
+
}
|
|
298
|
+
if let Some(texts) = whole_alternatives(pattern) {
|
|
299
|
+
return Some(Matcher::Literals(Literals::new(texts)));
|
|
300
|
+
}
|
|
301
|
+
if let Some(alts) = alternatives(pattern) {
|
|
302
|
+
return Some(Matcher::Alternatives(alts));
|
|
303
|
+
}
|
|
304
|
+
if let Some(list) = SeparatedList::parse(pattern) {
|
|
305
|
+
return Some(Matcher::SeparatedList(list));
|
|
306
|
+
}
|
|
307
|
+
if let Some((set, bangs)) = excluded_class_with_word(pattern) {
|
|
308
|
+
return Some(Matcher::ExcludedClassWithWord { set, bangs });
|
|
309
|
+
}
|
|
310
|
+
if let Some(seq) = Sequence::parse(pattern) {
|
|
311
|
+
return Some(match seq.literal() {
|
|
312
|
+
Some(text) => Matcher::Literal { text: text.into(), whole: seq.to_end },
|
|
313
|
+
None => Matcher::Sequence(seq),
|
|
314
|
+
});
|
|
315
|
+
}
|
|
316
|
+
let translated = translate(pattern, true)?;
|
|
317
|
+
regex::Regex::new(&translated).ok().map(Matcher::Regex)
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
/// `^.{m,n}$` (and `^.*$`, `^.+$`, `^.{m}$`, `^.{m,}$`, each also as `^(.…)$`): the bounds on the length of a line.
|
|
321
|
+
fn line_range(p: &str) -> Option<(u32, u32)> {
|
|
322
|
+
let q = p.strip_prefix("^.").or_else(|| p.strip_prefix("^(."))?.strip_suffix('$')?;
|
|
323
|
+
let q = if p.starts_with("^(") { q.strip_suffix(')')? } else { q };
|
|
324
|
+
let (min, max, next) = parse_quantifier(q.as_bytes(), 0)?;
|
|
325
|
+
(next == q.len() && next > 0 && !q.ends_with('?')).then_some((min, max))
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
/// Literal text: characters other than syntax characters, and identity or control escapes.
|
|
329
|
+
fn literal_text(p: &str) -> Option<String> {
|
|
330
|
+
let mut out = String::new();
|
|
331
|
+
let mut chars = p.chars();
|
|
332
|
+
while let Some(c) = chars.next() {
|
|
333
|
+
match c {
|
|
334
|
+
'\\' => out.push(match chars.next()? {
|
|
335
|
+
'n' => '\n',
|
|
336
|
+
'r' => '\r',
|
|
337
|
+
't' => '\t',
|
|
338
|
+
'f' => '\x0C',
|
|
339
|
+
'v' => '\x0B',
|
|
340
|
+
e
|
|
341
|
+
@ ('^' | '$' | '\\' | '.' | '*' | '+' | '?' | '(' | ')' | '[' | ']' | '{' | '}' | '|' | '/' | '-') => e,
|
|
342
|
+
_ => return None,
|
|
343
|
+
}),
|
|
344
|
+
'^' | '$' | '.' | '*' | '+' | '?' | '(' | ')' | '[' | ']' | '{' | '}' | '|' => return None,
|
|
345
|
+
c => out.push(c),
|
|
346
|
+
}
|
|
347
|
+
}
|
|
348
|
+
Some(out)
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
/// Splits at top-level `|`s (none inside a group, a class or after `\`).
|
|
352
|
+
fn split_alternatives(p: &str) -> Option<Vec<&str>> {
|
|
353
|
+
let b = p.as_bytes();
|
|
354
|
+
let (mut depth, mut in_class, mut i, mut start) = (0usize, false, 0, 0);
|
|
355
|
+
let mut parts = Vec::new();
|
|
356
|
+
while i < b.len() {
|
|
357
|
+
match b[i] {
|
|
358
|
+
b'\\' => i += 1,
|
|
359
|
+
b'[' => in_class = true,
|
|
360
|
+
b']' => in_class = false,
|
|
361
|
+
b'(' if !in_class => depth += 1,
|
|
362
|
+
b')' if !in_class => depth = depth.checked_sub(1)?,
|
|
363
|
+
b'|' if !in_class && depth == 0 => {
|
|
364
|
+
parts.push(&p[start..i]);
|
|
365
|
+
start = i + 1;
|
|
366
|
+
}
|
|
367
|
+
_ => {}
|
|
368
|
+
}
|
|
369
|
+
i += 1;
|
|
370
|
+
}
|
|
371
|
+
parts.push(&p[start..]);
|
|
372
|
+
Some(parts)
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
/// `^(a|b|...)$` or `^(?:a|b|...)$` over literal alternatives.
|
|
376
|
+
fn whole_alternatives(p: &str) -> Option<Vec<String>> {
|
|
377
|
+
let inner = p.strip_prefix("^(")?.strip_suffix(")$")?;
|
|
378
|
+
let inner = inner.strip_prefix("?:").unwrap_or(inner);
|
|
379
|
+
if inner.starts_with('?') {
|
|
380
|
+
return None;
|
|
381
|
+
}
|
|
382
|
+
let parts = split_alternatives(inner)?;
|
|
383
|
+
if parts.len() < 2 {
|
|
384
|
+
return None;
|
|
385
|
+
}
|
|
386
|
+
parts.into_iter().map(literal_text).collect()
|
|
387
|
+
}
|
|
388
|
+
|
|
389
|
+
/// At most this many alternatives after expanding groups.
|
|
390
|
+
const MAX_ALTERNATIVES: usize = 64;
|
|
391
|
+
|
|
392
|
+
/// At most this many alternatives when any is a class sequence: beyond it, trying each in turn is slower than the
|
|
393
|
+
/// `regex` crate's single automaton pass (measured on jsconfig's case-folded `lib` names).
|
|
394
|
+
const MAX_SEQUENCE_ALTERNATIVES: usize = 4;
|
|
395
|
+
|
|
396
|
+
/// Two or more alternatives once groups of alternatives (and optional groups) are expanded into whole alternatives
|
|
397
|
+
/// (`^a|b|c$`, `^([a|A]uto)|([n|N]one)$`, `^[Ee][Ss]2015(\.([Cc]ore|[Pp]roxy))?$`), each a literal or a class
|
|
398
|
+
/// sequence anchored where its own `^` and `$` say.
|
|
399
|
+
fn alternatives(p: &str) -> Option<Box<[Alternative]>> {
|
|
400
|
+
let c: Vec<char> = p.chars().collect();
|
|
401
|
+
let (parts, end) = expand(&c, 0)?;
|
|
402
|
+
// A single alternative is only new here when a group was expanded (Sequence::parse takes the rest).
|
|
403
|
+
if end != c.len() || parts.is_empty() || (parts.len() == 1 && parts[0] == p) {
|
|
404
|
+
return None;
|
|
405
|
+
}
|
|
406
|
+
let alternatives: Box<[Alternative]> = parts
|
|
407
|
+
.iter()
|
|
408
|
+
.map(|part| {
|
|
409
|
+
let (start, part) = part.strip_prefix('^').map_or((false, part.as_str()), |r| (true, r));
|
|
410
|
+
let (end, body) = match part.strip_suffix('$') {
|
|
411
|
+
Some(r) if !ends_with_escape(r) => (true, r),
|
|
412
|
+
_ => (false, part),
|
|
413
|
+
};
|
|
414
|
+
if let Some(text) = literal_text(body) {
|
|
415
|
+
return Some(Alternative::Literal { text: text.into(), start, end });
|
|
416
|
+
}
|
|
417
|
+
let seq = Sequence::parse(&format!("^{body}{}", if end { "$" } else { "" }))?;
|
|
418
|
+
if start {
|
|
419
|
+
return Some(Alternative::Start(seq));
|
|
420
|
+
}
|
|
421
|
+
let width =
|
|
422
|
+
seq.items.iter().all(|i| i.min == i.max).then(|| seq.items.iter().map(|i| i.min as usize).sum())?;
|
|
423
|
+
Some(if end { Alternative::End { seq, width } } else { Alternative::Anywhere { seq, width } })
|
|
424
|
+
})
|
|
425
|
+
.collect::<Option<_>>()?;
|
|
426
|
+
let sequences = alternatives.iter().any(|a| !matches!(a, Alternative::Literal { .. }));
|
|
427
|
+
(!sequences || alternatives.len() <= MAX_SEQUENCE_ALTERNATIVES).then_some(alternatives)
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
/// Whether the text ends in an unpaired `\` (so a `$` after it would be escaped).
|
|
431
|
+
fn ends_with_escape(t: &str) -> bool {
|
|
432
|
+
t.bytes().rev().take_while(|&b| b == b'\\').count() % 2 == 1
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
/// Expands the alternatives from `i` up to an unmatched `)` or the end: groups of alternatives multiply out, a group
|
|
436
|
+
/// quantified by `?` also contributes the empty alternative, and everything else is copied. `None` for lookarounds,
|
|
437
|
+
/// other quantified groups, or too many alternatives.
|
|
438
|
+
fn expand(c: &[char], mut i: usize) -> Option<(Vec<String>, usize)> {
|
|
439
|
+
let mut all = Vec::new();
|
|
440
|
+
let mut branch = vec![String::new()];
|
|
441
|
+
while i < c.len() {
|
|
442
|
+
match c[i] {
|
|
443
|
+
'|' => {
|
|
444
|
+
all.append(&mut branch);
|
|
445
|
+
branch.push(String::new());
|
|
446
|
+
i += 1;
|
|
447
|
+
}
|
|
448
|
+
')' => break,
|
|
449
|
+
'(' => {
|
|
450
|
+
i += 1;
|
|
451
|
+
if c.get(i) == Some(&'?') {
|
|
452
|
+
if c.get(i + 1) != Some(&':') {
|
|
453
|
+
return None;
|
|
454
|
+
}
|
|
455
|
+
i += 2;
|
|
456
|
+
}
|
|
457
|
+
let (mut inner, next) = expand(c, i)?;
|
|
458
|
+
if c.get(next) != Some(&')') {
|
|
459
|
+
return None;
|
|
460
|
+
}
|
|
461
|
+
i = next + 1;
|
|
462
|
+
match c.get(i) {
|
|
463
|
+
Some('?') => {
|
|
464
|
+
inner.push(String::new());
|
|
465
|
+
i += 1;
|
|
466
|
+
if c.get(i) == Some(&'?') {
|
|
467
|
+
i += 1;
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
Some('{') => {
|
|
471
|
+
// An exact count repeats the group; any other bound needs a real regex.
|
|
472
|
+
let close = i + c[i..].iter().position(|&x| x == '}')?;
|
|
473
|
+
let n: usize = c[i + 1..close].iter().collect::<String>().parse().ok()?;
|
|
474
|
+
if n > 16 {
|
|
475
|
+
return None;
|
|
476
|
+
}
|
|
477
|
+
let mut repeated = vec![String::new()];
|
|
478
|
+
for _ in 0..n {
|
|
479
|
+
if repeated.len() * inner.len() > MAX_ALTERNATIVES {
|
|
480
|
+
return None;
|
|
481
|
+
}
|
|
482
|
+
repeated =
|
|
483
|
+
repeated.iter().flat_map(|r| inner.iter().map(move |x| format!("{r}{x}"))).collect();
|
|
484
|
+
}
|
|
485
|
+
inner = repeated;
|
|
486
|
+
i = close + 1;
|
|
487
|
+
if c.get(i) == Some(&'?') {
|
|
488
|
+
i += 1;
|
|
489
|
+
}
|
|
490
|
+
}
|
|
491
|
+
Some('*' | '+') => return None,
|
|
492
|
+
_ => {}
|
|
493
|
+
}
|
|
494
|
+
if branch.len() * inner.len() > MAX_ALTERNATIVES {
|
|
495
|
+
return None;
|
|
496
|
+
}
|
|
497
|
+
branch = branch.iter().flat_map(|b| inner.iter().map(move |x| format!("{b}{x}"))).collect();
|
|
498
|
+
}
|
|
499
|
+
'[' => {
|
|
500
|
+
let start = i;
|
|
501
|
+
i += 1;
|
|
502
|
+
if c.get(i) == Some(&'^') {
|
|
503
|
+
i += 1;
|
|
504
|
+
}
|
|
505
|
+
if c.get(i) == Some(&']') {
|
|
506
|
+
i += 1;
|
|
507
|
+
}
|
|
508
|
+
while *c.get(i)? != ']' {
|
|
509
|
+
i += if c[i] == '\\' { 2 } else { 1 };
|
|
510
|
+
}
|
|
511
|
+
i += 1;
|
|
512
|
+
let text: String = c[start..i].iter().collect();
|
|
513
|
+
branch.iter_mut().for_each(|b| b.push_str(&text));
|
|
514
|
+
}
|
|
515
|
+
'\\' => {
|
|
516
|
+
let text: String = c.get(i..i + 2)?.iter().collect();
|
|
517
|
+
branch.iter_mut().for_each(|b| b.push_str(&text));
|
|
518
|
+
i += 2;
|
|
519
|
+
}
|
|
520
|
+
ch => {
|
|
521
|
+
branch.iter_mut().for_each(|b| b.push(ch));
|
|
522
|
+
i += 1;
|
|
523
|
+
}
|
|
524
|
+
}
|
|
525
|
+
if all.len() + branch.len() > MAX_ALTERNATIVES {
|
|
526
|
+
return None;
|
|
527
|
+
}
|
|
528
|
+
}
|
|
529
|
+
all.append(&mut branch);
|
|
530
|
+
Some((all, i))
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
/// `^(?=[^SET]+$)(?=(.*\w)).+$` (with `(?:` or `(` around `.*\w`): the excluded set.
|
|
534
|
+
fn excluded_class_with_word(p: &str) -> Option<(CharSet, bool)> {
|
|
535
|
+
// parse_class reads a class body from after `[`, here `^SET]`.
|
|
536
|
+
let (bangs, body) = match p.strip_prefix("^(?=!+[") {
|
|
537
|
+
Some(body) => (true, body),
|
|
538
|
+
None => (false, p.strip_prefix("^(?=[")?),
|
|
539
|
+
};
|
|
540
|
+
let (negated, next) = parse_class(body.as_bytes(), 0)?;
|
|
541
|
+
if !body.starts_with('^') {
|
|
542
|
+
return None;
|
|
543
|
+
}
|
|
544
|
+
let excluded = CharSet { ascii: !negated.ascii, ..CharSet::EMPTY };
|
|
545
|
+
// The run of `!` ends exactly where the class starts only when the class excludes `!`.
|
|
546
|
+
if bangs && !excluded.contains('!') {
|
|
547
|
+
return None;
|
|
548
|
+
}
|
|
549
|
+
matches!(&body[next..], "+$)(?=(.*\\w)).+$" | "+$)(?=(?:.*\\w)).+$" | "+$)(?=.*\\w).+$")
|
|
550
|
+
.then_some((excluded, bangs))
|
|
551
|
+
}
|
|
552
|
+
|
|
553
|
+
/// The `regex` format: a valid ECMA-262 regular expression (with the `u` flag).
|
|
554
|
+
pub(crate) fn is_valid_ecma_regex(s: &str) -> bool {
|
|
555
|
+
regress::Regex::with_flags(s, "u").is_ok()
|
|
556
|
+
}
|
|
557
|
+
|
|
558
|
+
// ---------------------------------------------------------------------------------------------------------------------
|
|
559
|
+
// Class sequences
|
|
560
|
+
|
|
561
|
+
/// A set of characters: ASCII by bitmask, then either all or none of the line separators U+2028 and U+2029, and
|
|
562
|
+
/// either all or none of the other non-ASCII characters.
|
|
563
|
+
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
|
|
564
|
+
struct CharSet {
|
|
565
|
+
ascii: u128,
|
|
566
|
+
non_ascii: bool,
|
|
567
|
+
separators: bool,
|
|
568
|
+
}
|
|
569
|
+
|
|
570
|
+
impl CharSet {
|
|
571
|
+
const EMPTY: CharSet = CharSet { ascii: 0, non_ascii: false, separators: false };
|
|
572
|
+
|
|
573
|
+
/// ASCII `lo..=hi` (`hi` below 128).
|
|
574
|
+
const fn range(lo: u8, hi: u8) -> CharSet {
|
|
575
|
+
let mut ascii = 0u128;
|
|
576
|
+
let mut c = lo;
|
|
577
|
+
while c <= hi {
|
|
578
|
+
ascii |= 1 << c;
|
|
579
|
+
c += 1;
|
|
580
|
+
}
|
|
581
|
+
CharSet { ascii, ..CharSet::EMPTY }
|
|
582
|
+
}
|
|
583
|
+
|
|
584
|
+
/// ECMA-262's `.`: everything but the line terminators.
|
|
585
|
+
fn dot() -> CharSet {
|
|
586
|
+
CharSet { ascii: !((1 << b'\n') | (1 << b'\r')), non_ascii: true, separators: false }
|
|
587
|
+
}
|
|
588
|
+
|
|
589
|
+
const fn union(self, other: CharSet) -> CharSet {
|
|
590
|
+
CharSet {
|
|
591
|
+
ascii: self.ascii | other.ascii,
|
|
592
|
+
non_ascii: self.non_ascii || other.non_ascii,
|
|
593
|
+
separators: self.separators || other.separators,
|
|
594
|
+
}
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
fn negate(self) -> CharSet {
|
|
598
|
+
CharSet { ascii: !self.ascii, non_ascii: !self.non_ascii, separators: !self.separators }
|
|
599
|
+
}
|
|
600
|
+
|
|
601
|
+
fn disjoint(self, other: CharSet) -> bool {
|
|
602
|
+
self.ascii & other.ascii == 0 && !(self.non_ascii && other.non_ascii) && !(self.separators && other.separators)
|
|
603
|
+
}
|
|
604
|
+
|
|
605
|
+
/// Whether the set holds an ASCII character (a `u128` bit test costs several instructions; a word select fewer).
|
|
606
|
+
#[inline]
|
|
607
|
+
fn has_ascii(&self, c: u8) -> bool {
|
|
608
|
+
let word = if c < 64 { self.ascii as u64 } else { (self.ascii >> 64) as u64 };
|
|
609
|
+
(word >> (c & 63)) & 1 != 0
|
|
610
|
+
}
|
|
611
|
+
|
|
612
|
+
#[inline]
|
|
613
|
+
fn contains(self, c: char) -> bool {
|
|
614
|
+
match c as u32 {
|
|
615
|
+
c if c < 128 => self.has_ascii(c as u8),
|
|
616
|
+
0x2028 | 0x2029 => self.separators,
|
|
617
|
+
_ => self.non_ascii,
|
|
618
|
+
}
|
|
619
|
+
}
|
|
620
|
+
|
|
621
|
+
const DIGIT: CharSet = CharSet::range(b'0', b'9');
|
|
622
|
+
|
|
623
|
+
/// `\w`. A constant: built per call, its ranges cost tens of nanoseconds, which matchers testing it per character
|
|
624
|
+
/// paid per character.
|
|
625
|
+
const WORD: CharSet = CharSet::range(b'0', b'9')
|
|
626
|
+
.union(CharSet::range(b'A', b'Z'))
|
|
627
|
+
.union(CharSet::range(b'a', b'z'))
|
|
628
|
+
.union(CharSet::range(b'_', b'_'));
|
|
629
|
+
|
|
630
|
+
fn digit() -> CharSet {
|
|
631
|
+
CharSet::DIGIT
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
fn word() -> CharSet {
|
|
635
|
+
CharSet::WORD
|
|
636
|
+
}
|
|
637
|
+
}
|
|
638
|
+
|
|
639
|
+
#[derive(Debug, Clone)]
|
|
640
|
+
struct Item {
|
|
641
|
+
set: CharSet,
|
|
642
|
+
min: u32,
|
|
643
|
+
/// `u32::MAX`: unbounded.
|
|
644
|
+
max: u32,
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
/// `^` then quantified character sets, optionally `$`. Matched greedily, which is exact because every variable item's
|
|
648
|
+
/// set is disjoint from the next item's (a character the item leaves cannot be taken by the next one either).
|
|
649
|
+
#[derive(Debug)]
|
|
650
|
+
struct Sequence {
|
|
651
|
+
items: Box<[Item]>,
|
|
652
|
+
to_end: bool,
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
impl Sequence {
|
|
656
|
+
fn parse(p: &str) -> Option<Sequence> {
|
|
657
|
+
let b = p.as_bytes();
|
|
658
|
+
if !p.is_ascii() || b.first() != Some(&b'^') {
|
|
659
|
+
return None;
|
|
660
|
+
}
|
|
661
|
+
let mut i = 1;
|
|
662
|
+
let mut items = Vec::new();
|
|
663
|
+
let mut to_end = false;
|
|
664
|
+
while i < b.len() {
|
|
665
|
+
let set = match b[i] {
|
|
666
|
+
b'$' if i == b.len() - 1 => {
|
|
667
|
+
to_end = true;
|
|
668
|
+
i += 1;
|
|
669
|
+
break;
|
|
670
|
+
}
|
|
671
|
+
b'[' => {
|
|
672
|
+
let (set, next) = parse_class(b, i + 1)?;
|
|
673
|
+
i = next;
|
|
674
|
+
set
|
|
675
|
+
}
|
|
676
|
+
b'\\' => {
|
|
677
|
+
let set = class_escape(*b.get(i + 1)?)?;
|
|
678
|
+
i += 2;
|
|
679
|
+
set
|
|
680
|
+
}
|
|
681
|
+
b'.' => {
|
|
682
|
+
i += 1;
|
|
683
|
+
CharSet::dot()
|
|
684
|
+
}
|
|
685
|
+
b'(' | b')' | b'|' | b'^' | b'$' | b'*' | b'+' | b'?' | b'{' | b'}' | b']' => return None,
|
|
686
|
+
c => {
|
|
687
|
+
i += 1;
|
|
688
|
+
CharSet::range(c, c)
|
|
689
|
+
}
|
|
690
|
+
};
|
|
691
|
+
let (min, max, next) = parse_quantifier(b, i)?;
|
|
692
|
+
i = next;
|
|
693
|
+
items.push(Item { set, min, max });
|
|
694
|
+
}
|
|
695
|
+
if i != b.len() {
|
|
696
|
+
return None;
|
|
697
|
+
}
|
|
698
|
+
// Greedy matching is exact only when a variable item cannot give up characters an item after it needs: its set
|
|
699
|
+
// must be disjoint from every item that can directly follow it (up to the first that cannot match nothing).
|
|
700
|
+
for (i, item) in items.iter().enumerate() {
|
|
701
|
+
if item.min == item.max {
|
|
702
|
+
continue;
|
|
703
|
+
}
|
|
704
|
+
for next in &items[i + 1..] {
|
|
705
|
+
if !item.set.disjoint(next.set) {
|
|
706
|
+
return None;
|
|
707
|
+
}
|
|
708
|
+
if next.min > 0 {
|
|
709
|
+
break;
|
|
710
|
+
}
|
|
711
|
+
}
|
|
712
|
+
}
|
|
713
|
+
Some(Sequence { items: items.into_boxed_slice(), to_end })
|
|
714
|
+
}
|
|
715
|
+
|
|
716
|
+
/// The text, when every item is one fixed character.
|
|
717
|
+
fn literal(&self) -> Option<String> {
|
|
718
|
+
self.items
|
|
719
|
+
.iter()
|
|
720
|
+
.map(|i| {
|
|
721
|
+
let single =
|
|
722
|
+
i.min == 1 && i.max == 1 && !i.set.non_ascii && !i.set.separators && i.set.ascii.count_ones() == 1;
|
|
723
|
+
single.then(|| i.set.ascii.trailing_zeros() as u8 as char)
|
|
724
|
+
})
|
|
725
|
+
.collect()
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
fn is_match(&self, s: &str) -> bool {
|
|
729
|
+
if s.is_ascii() { self.is_match_ascii(s.as_bytes()) } else { self.is_match_chars(s) }
|
|
730
|
+
}
|
|
731
|
+
|
|
732
|
+
/// The same over any text, a character at a time.
|
|
733
|
+
fn is_match_chars(&self, s: &str) -> bool {
|
|
734
|
+
let mut chars = s.chars().peekable();
|
|
735
|
+
for item in self.items.iter() {
|
|
736
|
+
let mut n = 0u32;
|
|
737
|
+
while n < item.max {
|
|
738
|
+
match chars.peek() {
|
|
739
|
+
Some(&c) if item.set.contains(c) => {
|
|
740
|
+
chars.next();
|
|
741
|
+
n += 1;
|
|
742
|
+
}
|
|
743
|
+
_ => break,
|
|
744
|
+
}
|
|
745
|
+
}
|
|
746
|
+
if n < item.min {
|
|
747
|
+
return false;
|
|
748
|
+
}
|
|
749
|
+
}
|
|
750
|
+
!self.to_end || chars.next().is_none()
|
|
751
|
+
}
|
|
752
|
+
|
|
753
|
+
/// The same over ASCII text, a byte per character.
|
|
754
|
+
fn is_match_ascii(&self, b: &[u8]) -> bool {
|
|
755
|
+
let mut at = 0usize;
|
|
756
|
+
for item in self.items.iter() {
|
|
757
|
+
let start = at;
|
|
758
|
+
let end = b.len().min(start.saturating_add(item.max as usize));
|
|
759
|
+
while at < end && item.set.has_ascii(b[at]) {
|
|
760
|
+
at += 1;
|
|
761
|
+
}
|
|
762
|
+
if at - start < item.min as usize {
|
|
763
|
+
return false;
|
|
764
|
+
}
|
|
765
|
+
}
|
|
766
|
+
!self.to_end || at == b.len()
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
/// Greedily matches the items at the start of `s` (ignoring `to_end`): the bytes taken.
|
|
770
|
+
fn consume(&self, s: &str) -> Option<usize> {
|
|
771
|
+
let b = s.as_bytes();
|
|
772
|
+
let mut at = 0;
|
|
773
|
+
for item in self.items.iter() {
|
|
774
|
+
let mut n = 0u32;
|
|
775
|
+
while n < item.max && at < b.len() {
|
|
776
|
+
let c = b[at];
|
|
777
|
+
if c < 0x80 {
|
|
778
|
+
if !item.set.has_ascii(c) {
|
|
779
|
+
break;
|
|
780
|
+
}
|
|
781
|
+
at += 1;
|
|
782
|
+
} else {
|
|
783
|
+
let ch = s[at..].chars().next()?;
|
|
784
|
+
if !item.set.contains(ch) {
|
|
785
|
+
break;
|
|
786
|
+
}
|
|
787
|
+
at += ch.len_utf8();
|
|
788
|
+
}
|
|
789
|
+
n += 1;
|
|
790
|
+
}
|
|
791
|
+
if n < item.min {
|
|
792
|
+
return None;
|
|
793
|
+
}
|
|
794
|
+
}
|
|
795
|
+
Some(at)
|
|
796
|
+
}
|
|
797
|
+
|
|
798
|
+
/// Whether greedy matching is exact when `next` can follow the items: every variable item is disjoint from the
|
|
799
|
+
/// items that can directly follow it (up to the first that cannot match nothing), `next` included.
|
|
800
|
+
fn greedy_before(items: &[Item], next: CharSet) -> bool {
|
|
801
|
+
items.iter().enumerate().all(|(i, item)| {
|
|
802
|
+
if item.min == item.max {
|
|
803
|
+
return true;
|
|
804
|
+
}
|
|
805
|
+
for following in &items[i + 1..] {
|
|
806
|
+
if !item.set.disjoint(following.set) {
|
|
807
|
+
return false;
|
|
808
|
+
}
|
|
809
|
+
if following.min > 0 {
|
|
810
|
+
return true;
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
item.set.disjoint(next)
|
|
814
|
+
})
|
|
815
|
+
}
|
|
816
|
+
}
|
|
817
|
+
|
|
818
|
+
/// A list of items between separators: `^I(SR)*$` or `^I(SR)+$` (`final_` is `None`), or `^(RS)*F$` and
|
|
819
|
+
/// `^(RS)+F$`. The separator is a fixed sequence and no variable class can run into what follows it, so greedy
|
|
820
|
+
/// matching splits the string exactly where the pattern does.
|
|
821
|
+
#[derive(Debug)]
|
|
822
|
+
struct SeparatedList {
|
|
823
|
+
/// `I` of the first form.
|
|
824
|
+
first: Option<Sequence>,
|
|
825
|
+
repeated: Sequence,
|
|
826
|
+
separator: Sequence,
|
|
827
|
+
/// `F` of the second form, matched against the whole remainder.
|
|
828
|
+
final_: Option<Sequence>,
|
|
829
|
+
min_repeats: u32,
|
|
830
|
+
}
|
|
831
|
+
|
|
832
|
+
impl SeparatedList {
|
|
833
|
+
fn is_match(&self, s: &str) -> bool {
|
|
834
|
+
let mut rest = s;
|
|
835
|
+
let mut repeats = 0;
|
|
836
|
+
if let Some(first) = &self.first {
|
|
837
|
+
let Some(n) = first.consume(rest) else { return false };
|
|
838
|
+
rest = &rest[n..];
|
|
839
|
+
loop {
|
|
840
|
+
if rest.is_empty() {
|
|
841
|
+
return repeats >= self.min_repeats;
|
|
842
|
+
}
|
|
843
|
+
let Some(n) = self.separator.consume(rest) else { return false };
|
|
844
|
+
rest = &rest[n..];
|
|
845
|
+
let Some(n) = self.repeated.consume(rest) else { return false };
|
|
846
|
+
rest = &rest[n..];
|
|
847
|
+
repeats += 1;
|
|
848
|
+
}
|
|
849
|
+
}
|
|
850
|
+
let final_ = self.final_.as_ref().expect("a separated list has a first or a final item");
|
|
851
|
+
loop {
|
|
852
|
+
if repeats >= self.min_repeats && final_.is_match(rest) {
|
|
853
|
+
return true;
|
|
854
|
+
}
|
|
855
|
+
let Some(n) = self.repeated.consume(rest) else { return false };
|
|
856
|
+
let Some(m) = self.separator.consume(&rest[n..]) else { return false };
|
|
857
|
+
rest = &rest[n + m..];
|
|
858
|
+
repeats += 1;
|
|
859
|
+
}
|
|
860
|
+
}
|
|
861
|
+
|
|
862
|
+
fn parse(p: &str) -> Option<SeparatedList> {
|
|
863
|
+
let body = p.strip_prefix('^')?.strip_suffix('$')?;
|
|
864
|
+
if ends_with_escape(body) {
|
|
865
|
+
return None;
|
|
866
|
+
}
|
|
867
|
+
let b = body.as_bytes();
|
|
868
|
+
// Top-level groups: (start, end) of each, where end is the index of its `)`.
|
|
869
|
+
let mut groups = Vec::new();
|
|
870
|
+
let (mut depth, mut in_class, mut i, mut open) = (0usize, false, 0, 0);
|
|
871
|
+
while i < b.len() {
|
|
872
|
+
match b[i] {
|
|
873
|
+
b'\\' => i += 1,
|
|
874
|
+
b'[' if !in_class => in_class = true,
|
|
875
|
+
b']' if in_class => in_class = false,
|
|
876
|
+
b'(' if !in_class => {
|
|
877
|
+
if depth == 0 {
|
|
878
|
+
open = i;
|
|
879
|
+
}
|
|
880
|
+
depth += 1;
|
|
881
|
+
}
|
|
882
|
+
b')' if !in_class => {
|
|
883
|
+
depth = depth.checked_sub(1)?;
|
|
884
|
+
if depth == 0 {
|
|
885
|
+
groups.push((open, i));
|
|
886
|
+
}
|
|
887
|
+
}
|
|
888
|
+
_ => {}
|
|
889
|
+
}
|
|
890
|
+
i += 1;
|
|
891
|
+
}
|
|
892
|
+
let quantified: Vec<_> = groups.iter().filter(|&&(_, e)| matches!(b.get(e + 1), Some(b'*' | b'+'))).collect();
|
|
893
|
+
let &&(open, close) = quantified.first().filter(|_| quantified.len() == 1)?;
|
|
894
|
+
let min_repeats = u32::from(b[close + 1] == b'+');
|
|
895
|
+
let group = strip_group(&body[open..=close])?;
|
|
896
|
+
let before = &body[..open];
|
|
897
|
+
let after = &body[close + 2..];
|
|
898
|
+
let items = |t: &str| Sequence::parse(&format!("^{t}")).map(|s| s.items.into_vec());
|
|
899
|
+
let seq = |items: &[Item], to_end| Sequence { items: items.into(), to_end };
|
|
900
|
+
let fixed = |items: &[Item]| !items.is_empty() && items.iter().all(|i| i.min == i.max && i.min > 0);
|
|
901
|
+
let g = items(group)?;
|
|
902
|
+
match (before.is_empty(), after.is_empty()) {
|
|
903
|
+
// ^I(SR)*$
|
|
904
|
+
(false, true) => {
|
|
905
|
+
let first = items(unwrap_group(before)?)?;
|
|
906
|
+
(1..g.len()).find_map(|k| {
|
|
907
|
+
let (separator, repeated) = g.split_at(k);
|
|
908
|
+
let next = separator[0].set;
|
|
909
|
+
(fixed(separator)
|
|
910
|
+
&& Sequence::greedy_before(&first, next)
|
|
911
|
+
&& Sequence::greedy_before(repeated, next))
|
|
912
|
+
.then(|| SeparatedList {
|
|
913
|
+
first: Some(seq(&first, false)),
|
|
914
|
+
repeated: seq(repeated, false),
|
|
915
|
+
separator: seq(separator, false),
|
|
916
|
+
final_: None,
|
|
917
|
+
min_repeats,
|
|
918
|
+
})
|
|
919
|
+
})
|
|
920
|
+
}
|
|
921
|
+
// ^(RS)*F$
|
|
922
|
+
(true, false) => {
|
|
923
|
+
let final_ = Sequence::parse(&format!("^{}$", unwrap_group(after)?))?;
|
|
924
|
+
(1..g.len()).find_map(|k| {
|
|
925
|
+
let (repeated, separator) = g.split_at(k);
|
|
926
|
+
(fixed(separator) && Sequence::greedy_before(repeated, separator[0].set)).then(|| SeparatedList {
|
|
927
|
+
first: None,
|
|
928
|
+
repeated: seq(repeated, false),
|
|
929
|
+
separator: seq(separator, false),
|
|
930
|
+
final_: Some(Sequence { items: final_.items.clone(), to_end: true }),
|
|
931
|
+
min_repeats,
|
|
932
|
+
})
|
|
933
|
+
})
|
|
934
|
+
}
|
|
935
|
+
_ => None,
|
|
936
|
+
}
|
|
937
|
+
}
|
|
938
|
+
}
|
|
939
|
+
|
|
940
|
+
/// The inside of `(...)` or `(?:...)`; `None` for other groups.
|
|
941
|
+
fn strip_group(g: &str) -> Option<&str> {
|
|
942
|
+
let inner = g.strip_prefix('(')?.strip_suffix(')')?;
|
|
943
|
+
match inner.strip_prefix("?:") {
|
|
944
|
+
Some(rest) => Some(rest),
|
|
945
|
+
None if inner.starts_with('?') => None,
|
|
946
|
+
None => Some(inner),
|
|
947
|
+
}
|
|
948
|
+
}
|
|
949
|
+
|
|
950
|
+
/// A text that is one group wrapping everything, unwrapped; otherwise the text itself.
|
|
951
|
+
fn unwrap_group(t: &str) -> Option<&str> {
|
|
952
|
+
if t.starts_with('(') && t.ends_with(')') {
|
|
953
|
+
let inner = strip_group(t)?;
|
|
954
|
+
// Only when the parentheses enclose the whole text (`(a)(b)` is two groups).
|
|
955
|
+
let mut depth = 0i32;
|
|
956
|
+
let mut escaped = false;
|
|
957
|
+
for c in inner.chars() {
|
|
958
|
+
match c {
|
|
959
|
+
_ if escaped => escaped = false,
|
|
960
|
+
'\\' => escaped = true,
|
|
961
|
+
'(' => depth += 1,
|
|
962
|
+
')' => {
|
|
963
|
+
depth -= 1;
|
|
964
|
+
if depth < 0 {
|
|
965
|
+
return None;
|
|
966
|
+
}
|
|
967
|
+
}
|
|
968
|
+
_ => {}
|
|
969
|
+
}
|
|
970
|
+
}
|
|
971
|
+
return Some(inner);
|
|
972
|
+
}
|
|
973
|
+
Some(t)
|
|
974
|
+
}
|
|
975
|
+
|
|
976
|
+
/// The set of a class escape (`\d`, `\w`, their negations, or an escaped punctuation character); `None` for `\s`
|
|
977
|
+
/// (not ASCII-only) and anything else.
|
|
978
|
+
fn class_escape(c: u8) -> Option<CharSet> {
|
|
979
|
+
Some(match c {
|
|
980
|
+
b'd' => CharSet::digit(),
|
|
981
|
+
b'D' => CharSet::digit().negate(),
|
|
982
|
+
b'w' => CharSet::word(),
|
|
983
|
+
b'W' => CharSet::word().negate(),
|
|
984
|
+
b'n' => CharSet::range(b'\n', b'\n'),
|
|
985
|
+
b'r' => CharSet::range(b'\r', b'\r'),
|
|
986
|
+
b't' => CharSet::range(b'\t', b'\t'),
|
|
987
|
+
c if c.is_ascii_punctuation() => CharSet::range(c, c),
|
|
988
|
+
_ => return None,
|
|
989
|
+
})
|
|
990
|
+
}
|
|
991
|
+
|
|
992
|
+
/// A class body from after `[` to after `]`, with ASCII members (a negated class also takes every non-ASCII
|
|
993
|
+
/// character).
|
|
994
|
+
fn parse_class(b: &[u8], mut i: usize) -> Option<(CharSet, usize)> {
|
|
995
|
+
let negated = b.get(i) == Some(&b'^');
|
|
996
|
+
if negated {
|
|
997
|
+
i += 1;
|
|
998
|
+
}
|
|
999
|
+
let mut set = CharSet::EMPTY;
|
|
1000
|
+
let mut first = true;
|
|
1001
|
+
loop {
|
|
1002
|
+
let c = *b.get(i)?;
|
|
1003
|
+
if c == b']' && !first {
|
|
1004
|
+
i += 1;
|
|
1005
|
+
break;
|
|
1006
|
+
}
|
|
1007
|
+
if c == b']' {
|
|
1008
|
+
return None; // `[]` / `[^]`
|
|
1009
|
+
}
|
|
1010
|
+
first = false;
|
|
1011
|
+
// One atom: a single character (for ranges) or a set escape.
|
|
1012
|
+
let (atom, single, next) = match c {
|
|
1013
|
+
b'\\' => {
|
|
1014
|
+
let e = *b.get(i + 1)?;
|
|
1015
|
+
let s = class_escape(e)?;
|
|
1016
|
+
let single = matches!(e, b'n' | b'r' | b't') || e.is_ascii_punctuation();
|
|
1017
|
+
(s, single.then(|| s.ascii.trailing_zeros() as u8), i + 2)
|
|
1018
|
+
}
|
|
1019
|
+
_ => (CharSet::range(c, c), Some(c), i + 1),
|
|
1020
|
+
};
|
|
1021
|
+
i = next;
|
|
1022
|
+
if b.get(i) == Some(&b'-') && b.get(i + 1).is_some_and(|&n| n != b']') {
|
|
1023
|
+
let lo = single?;
|
|
1024
|
+
let hi = match b[i + 1] {
|
|
1025
|
+
b'\\' => {
|
|
1026
|
+
let e = *b.get(i + 2)?;
|
|
1027
|
+
if !e.is_ascii_punctuation() {
|
|
1028
|
+
return None;
|
|
1029
|
+
}
|
|
1030
|
+
i += 3;
|
|
1031
|
+
e
|
|
1032
|
+
}
|
|
1033
|
+
h => {
|
|
1034
|
+
i += 2;
|
|
1035
|
+
h
|
|
1036
|
+
}
|
|
1037
|
+
};
|
|
1038
|
+
if lo > hi {
|
|
1039
|
+
return None;
|
|
1040
|
+
}
|
|
1041
|
+
set = set.union(CharSet::range(lo, hi));
|
|
1042
|
+
} else {
|
|
1043
|
+
set = set.union(atom);
|
|
1044
|
+
}
|
|
1045
|
+
}
|
|
1046
|
+
Some((if negated { set.negate() } else { set }, i))
|
|
1047
|
+
}
|
|
1048
|
+
|
|
1049
|
+
/// An optional quantifier (`*`, `+`, `?`, `{n}`, `{n,}`, `{n,m}`, each optionally lazy) at `i`.
|
|
1050
|
+
fn parse_quantifier(b: &[u8], mut i: usize) -> Option<(u32, u32, usize)> {
|
|
1051
|
+
let (min, max) = match b.get(i) {
|
|
1052
|
+
Some(b'*') => {
|
|
1053
|
+
i += 1;
|
|
1054
|
+
(0, u32::MAX)
|
|
1055
|
+
}
|
|
1056
|
+
Some(b'+') => {
|
|
1057
|
+
i += 1;
|
|
1058
|
+
(1, u32::MAX)
|
|
1059
|
+
}
|
|
1060
|
+
Some(b'?') => {
|
|
1061
|
+
i += 1;
|
|
1062
|
+
(0, 1)
|
|
1063
|
+
}
|
|
1064
|
+
Some(b'{') => {
|
|
1065
|
+
let end = i + b[i..].iter().position(|&c| c == b'}')?;
|
|
1066
|
+
let body = std::str::from_utf8(&b[i + 1..end]).ok()?;
|
|
1067
|
+
i = end + 1;
|
|
1068
|
+
match body.split_once(',') {
|
|
1069
|
+
None => {
|
|
1070
|
+
let n = body.parse().ok()?;
|
|
1071
|
+
(n, n)
|
|
1072
|
+
}
|
|
1073
|
+
Some((lo, "")) => (lo.parse().ok()?, u32::MAX),
|
|
1074
|
+
Some((lo, hi)) => (lo.parse().ok()?, hi.parse().ok()?),
|
|
1075
|
+
}
|
|
1076
|
+
}
|
|
1077
|
+
_ => return Some((1, 1, i)),
|
|
1078
|
+
};
|
|
1079
|
+
if min > max {
|
|
1080
|
+
return None;
|
|
1081
|
+
}
|
|
1082
|
+
if b.get(i) == Some(&b'?') {
|
|
1083
|
+
i += 1; // Laziness does not change whether the whole pattern matches.
|
|
1084
|
+
}
|
|
1085
|
+
Some((min, max, i))
|
|
1086
|
+
}
|
|
1087
|
+
|
|
1088
|
+
// ---------------------------------------------------------------------------------------------------------------------
|
|
1089
|
+
// Translation to the regex crate
|
|
1090
|
+
|
|
1091
|
+
/// ECMA-262 `\s`: WhiteSpace and LineTerminator.
|
|
1092
|
+
const ECMA_SPACE: &str =
|
|
1093
|
+
r"\t\n\x{B}\x{C}\r \x{A0}\x{1680}\x{2000}-\x{200A}\x{2028}\x{2029}\x{202F}\x{205F}\x{3000}\x{FEFF}";
|
|
1094
|
+
|
|
1095
|
+
/// Translates an ECMA-262 pattern to the `regex` crate's syntax with the same meaning, or `None` when it uses
|
|
1096
|
+
/// something outside the common subset. With `unicode` false, the pattern is read without the `u` flag (identity
|
|
1097
|
+
/// escapes of any non-alphanumeric character), and the translation has its meaning on strings within the BMP.
|
|
1098
|
+
fn translate(p: &str, unicode: bool) -> Option<String> {
|
|
1099
|
+
let c: Vec<char> = p.chars().collect();
|
|
1100
|
+
let mut out = String::with_capacity(p.len() * 2);
|
|
1101
|
+
let mut i = 0;
|
|
1102
|
+
while i < c.len() {
|
|
1103
|
+
match c[i] {
|
|
1104
|
+
'\\' => {
|
|
1105
|
+
let (text, next) = escape(&c, i + 1, false, unicode)?;
|
|
1106
|
+
out.push_str(&text);
|
|
1107
|
+
i = next;
|
|
1108
|
+
}
|
|
1109
|
+
'[' => {
|
|
1110
|
+
let (text, next) = class(&c, i + 1, unicode)?;
|
|
1111
|
+
out.push_str(&text);
|
|
1112
|
+
i = next;
|
|
1113
|
+
}
|
|
1114
|
+
'(' => {
|
|
1115
|
+
if c.get(i + 1) == Some(&'?') {
|
|
1116
|
+
match (c.get(i + 2), c.get(i + 3)) {
|
|
1117
|
+
(Some(':'), _) => i += 3,
|
|
1118
|
+
// A named group; lookbehinds (`(?<=`, `(?<!`) are not supported.
|
|
1119
|
+
(Some('<'), Some(n)) if *n != '=' && *n != '!' => {
|
|
1120
|
+
i += 3 + c[i + 3..].iter().position(|&x| x == '>')? + 1;
|
|
1121
|
+
}
|
|
1122
|
+
_ => return None,
|
|
1123
|
+
}
|
|
1124
|
+
} else {
|
|
1125
|
+
i += 1;
|
|
1126
|
+
}
|
|
1127
|
+
out.push_str("(?:");
|
|
1128
|
+
}
|
|
1129
|
+
'{' => {
|
|
1130
|
+
let end = i + c[i..].iter().position(|&x| x == '}')?;
|
|
1131
|
+
let body: String = c[i + 1..end].iter().collect();
|
|
1132
|
+
if body.is_empty() || !body.chars().all(|x| x.is_ascii_digit() || x == ',') {
|
|
1133
|
+
return None;
|
|
1134
|
+
}
|
|
1135
|
+
out.push('{');
|
|
1136
|
+
out.push_str(&body);
|
|
1137
|
+
out.push('}');
|
|
1138
|
+
i = end + 1;
|
|
1139
|
+
}
|
|
1140
|
+
'.' => {
|
|
1141
|
+
out.push_str(r"[^\n\r\x{2028}\x{2029}]");
|
|
1142
|
+
i += 1;
|
|
1143
|
+
}
|
|
1144
|
+
ch @ (')' | '|' | '^' | '$' | '*' | '+' | '?') => {
|
|
1145
|
+
out.push(ch);
|
|
1146
|
+
i += 1;
|
|
1147
|
+
}
|
|
1148
|
+
ch => {
|
|
1149
|
+
push_literal(&mut out, ch);
|
|
1150
|
+
i += 1;
|
|
1151
|
+
}
|
|
1152
|
+
}
|
|
1153
|
+
}
|
|
1154
|
+
Some(out)
|
|
1155
|
+
}
|
|
1156
|
+
|
|
1157
|
+
fn push_literal(out: &mut String, ch: char) {
|
|
1158
|
+
use std::fmt::Write;
|
|
1159
|
+
let _ = write!(out, "\\x{{{:X}}}", ch as u32);
|
|
1160
|
+
}
|
|
1161
|
+
|
|
1162
|
+
/// An escape after `\` at `i`: its translation and the index after it.
|
|
1163
|
+
fn escape(c: &[char], i: usize, in_class: bool, unicode: bool) -> Option<(String, usize)> {
|
|
1164
|
+
let e = *c.get(i)?;
|
|
1165
|
+
let mut out = String::new();
|
|
1166
|
+
let next = match e {
|
|
1167
|
+
'd' => {
|
|
1168
|
+
out.push_str(if in_class { "0-9" } else { "[0-9]" });
|
|
1169
|
+
i + 1
|
|
1170
|
+
}
|
|
1171
|
+
'D' => {
|
|
1172
|
+
out.push_str("[^0-9]");
|
|
1173
|
+
i + 1
|
|
1174
|
+
}
|
|
1175
|
+
'w' => {
|
|
1176
|
+
out.push_str(if in_class { "0-9A-Za-z_" } else { "[0-9A-Za-z_]" });
|
|
1177
|
+
i + 1
|
|
1178
|
+
}
|
|
1179
|
+
'W' => {
|
|
1180
|
+
out.push_str("[^0-9A-Za-z_]");
|
|
1181
|
+
i + 1
|
|
1182
|
+
}
|
|
1183
|
+
's' => {
|
|
1184
|
+
if in_class {
|
|
1185
|
+
out.push_str(ECMA_SPACE);
|
|
1186
|
+
} else {
|
|
1187
|
+
out.push('[');
|
|
1188
|
+
out.push_str(ECMA_SPACE);
|
|
1189
|
+
out.push(']');
|
|
1190
|
+
}
|
|
1191
|
+
i + 1
|
|
1192
|
+
}
|
|
1193
|
+
'S' => {
|
|
1194
|
+
out.push_str("[^");
|
|
1195
|
+
out.push_str(ECMA_SPACE);
|
|
1196
|
+
out.push(']');
|
|
1197
|
+
i + 1
|
|
1198
|
+
}
|
|
1199
|
+
'n' | 'r' | 't' | 'f' | 'v' => {
|
|
1200
|
+
let ch = match e {
|
|
1201
|
+
'n' => '\n',
|
|
1202
|
+
'r' => '\r',
|
|
1203
|
+
't' => '\t',
|
|
1204
|
+
'f' => '\x0C',
|
|
1205
|
+
_ => '\x0B',
|
|
1206
|
+
};
|
|
1207
|
+
push_literal(&mut out, ch);
|
|
1208
|
+
i + 1
|
|
1209
|
+
}
|
|
1210
|
+
'0' if !c.get(i + 1).is_some_and(char::is_ascii_digit) => {
|
|
1211
|
+
push_literal(&mut out, '\0');
|
|
1212
|
+
i + 1
|
|
1213
|
+
}
|
|
1214
|
+
'b' if in_class => {
|
|
1215
|
+
push_literal(&mut out, '\x08');
|
|
1216
|
+
i + 1
|
|
1217
|
+
}
|
|
1218
|
+
// ECMA's word boundaries are between ASCII word characters and the rest.
|
|
1219
|
+
'b' => {
|
|
1220
|
+
out.push_str("(?-u:\\b)");
|
|
1221
|
+
i + 1
|
|
1222
|
+
}
|
|
1223
|
+
'B' => {
|
|
1224
|
+
out.push_str("(?-u:\\B)");
|
|
1225
|
+
i + 1
|
|
1226
|
+
}
|
|
1227
|
+
// Unicode properties (validated as ECMA-262 names by regress) have the same names in the `regex` crate.
|
|
1228
|
+
// Without the `u` flag `\p` is the letter.
|
|
1229
|
+
'p' | 'P' if unicode && c.get(i + 1) == Some(&'{') => {
|
|
1230
|
+
let end = i + 1 + c[i + 1..].iter().position(|&x| x == '}')?;
|
|
1231
|
+
let body: String = c[i + 2..end].iter().collect();
|
|
1232
|
+
if body.is_empty() || !body.chars().all(|x| x.is_ascii_alphanumeric() || x == '_' || x == '=') {
|
|
1233
|
+
return None;
|
|
1234
|
+
}
|
|
1235
|
+
out.push('\\');
|
|
1236
|
+
out.push(e);
|
|
1237
|
+
out.push('{');
|
|
1238
|
+
out.push_str(&body);
|
|
1239
|
+
out.push('}');
|
|
1240
|
+
end + 1
|
|
1241
|
+
}
|
|
1242
|
+
'x' => {
|
|
1243
|
+
let hex: String = c.get(i + 1..i + 3)?.iter().collect();
|
|
1244
|
+
push_literal(&mut out, char::from_u32(u32::from_str_radix(&hex, 16).ok()?)?);
|
|
1245
|
+
i + 3
|
|
1246
|
+
}
|
|
1247
|
+
// Without the `u` flag, `\u{...}` is not a code point escape.
|
|
1248
|
+
'u' if unicode || c.get(i + 1) != Some(&'{') => {
|
|
1249
|
+
let (code, next) = unicode_escape(c, i + 1)?;
|
|
1250
|
+
push_literal(&mut out, char::from_u32(code)?);
|
|
1251
|
+
next
|
|
1252
|
+
}
|
|
1253
|
+
// Identity escapes of syntax characters and `/` (and `-` in a class).
|
|
1254
|
+
'^' | '$' | '\\' | '.' | '*' | '+' | '?' | '(' | ')' | '[' | ']' | '{' | '}' | '|' | '/' | '-' => {
|
|
1255
|
+
push_literal(&mut out, e);
|
|
1256
|
+
i + 1
|
|
1257
|
+
}
|
|
1258
|
+
// Without the `u` flag, any other non-alphanumeric character escapes to itself.
|
|
1259
|
+
e if !unicode && !e.is_ascii_alphanumeric() => {
|
|
1260
|
+
push_literal(&mut out, e);
|
|
1261
|
+
i + 1
|
|
1262
|
+
}
|
|
1263
|
+
// Backreferences, \c, \k and the rest.
|
|
1264
|
+
_ => return None,
|
|
1265
|
+
};
|
|
1266
|
+
Some((out, next))
|
|
1267
|
+
}
|
|
1268
|
+
|
|
1269
|
+
/// `\u` escapes after the `u`: `XXXX` (joining a surrogate pair) or `{X...}`.
|
|
1270
|
+
fn unicode_escape(c: &[char], i: usize) -> Option<(u32, usize)> {
|
|
1271
|
+
if c.get(i) == Some(&'{') {
|
|
1272
|
+
let end = i + c[i..].iter().position(|&x| x == '}')?;
|
|
1273
|
+
let hex: String = c[i + 1..end].iter().collect();
|
|
1274
|
+
return Some((u32::from_str_radix(&hex, 16).ok()?, end + 1));
|
|
1275
|
+
}
|
|
1276
|
+
let four = |at: usize| -> Option<u32> {
|
|
1277
|
+
let hex: String = c.get(at..at + 4)?.iter().collect();
|
|
1278
|
+
if hex.len() != 4 {
|
|
1279
|
+
return None;
|
|
1280
|
+
}
|
|
1281
|
+
u32::from_str_radix(&hex, 16).ok()
|
|
1282
|
+
};
|
|
1283
|
+
let hi = four(i)?;
|
|
1284
|
+
if (0xD800..0xDC00).contains(&hi) {
|
|
1285
|
+
if c.get(i + 4) == Some(&'\\') && c.get(i + 5) == Some(&'u') {
|
|
1286
|
+
let lo = four(i + 6)?;
|
|
1287
|
+
if (0xDC00..0xE000).contains(&lo) {
|
|
1288
|
+
return Some((0x10000 + ((hi - 0xD800) << 10) + (lo - 0xDC00), i + 10));
|
|
1289
|
+
}
|
|
1290
|
+
}
|
|
1291
|
+
return None;
|
|
1292
|
+
}
|
|
1293
|
+
Some((hi, i + 4))
|
|
1294
|
+
}
|
|
1295
|
+
|
|
1296
|
+
/// A class after `[` at `i`: its translation (every member spelled as an escape, so nothing in it is special to the
|
|
1297
|
+
/// `regex` crate) and the index after `]`.
|
|
1298
|
+
fn class(c: &[char], mut i: usize, unicode: bool) -> Option<(String, usize)> {
|
|
1299
|
+
let mut out = String::from("[");
|
|
1300
|
+
if c.get(i) == Some(&'^') {
|
|
1301
|
+
out.push('^');
|
|
1302
|
+
i += 1;
|
|
1303
|
+
}
|
|
1304
|
+
let mut members = 0;
|
|
1305
|
+
loop {
|
|
1306
|
+
let ch = *c.get(i)?;
|
|
1307
|
+
if ch == ']' {
|
|
1308
|
+
i += 1;
|
|
1309
|
+
break;
|
|
1310
|
+
}
|
|
1311
|
+
// One atom: a character (which can start a range) or a class escape.
|
|
1312
|
+
let (atom, single, next) = if ch == '\\' {
|
|
1313
|
+
let e = *c.get(i + 1)?;
|
|
1314
|
+
let (text, next) = escape(c, i + 1, true, unicode)?;
|
|
1315
|
+
let single = match e {
|
|
1316
|
+
'd' | 'D' | 'w' | 'W' | 's' | 'S' | 'p' | 'P' => None,
|
|
1317
|
+
_ => Some(text.clone()),
|
|
1318
|
+
};
|
|
1319
|
+
(text, single, next)
|
|
1320
|
+
} else {
|
|
1321
|
+
let mut t = String::new();
|
|
1322
|
+
push_literal(&mut t, ch);
|
|
1323
|
+
(t.clone(), Some(t), i + 1)
|
|
1324
|
+
};
|
|
1325
|
+
i = next;
|
|
1326
|
+
members += 1;
|
|
1327
|
+
if c.get(i) == Some(&'-') && c.get(i + 1).is_some_and(|&n| n != ']') {
|
|
1328
|
+
// A range: both ends single characters.
|
|
1329
|
+
let lo = single?;
|
|
1330
|
+
let (hi, next) = if c[i + 1] == '\\' {
|
|
1331
|
+
let e = *c.get(i + 2)?;
|
|
1332
|
+
if matches!(e, 'd' | 'D' | 'w' | 'W' | 's' | 'S' | 'p' | 'P') {
|
|
1333
|
+
return None;
|
|
1334
|
+
}
|
|
1335
|
+
escape(c, i + 2, true, unicode)?
|
|
1336
|
+
} else {
|
|
1337
|
+
let mut t = String::new();
|
|
1338
|
+
push_literal(&mut t, c[i + 1]);
|
|
1339
|
+
(t, i + 2)
|
|
1340
|
+
};
|
|
1341
|
+
out.push_str(&lo);
|
|
1342
|
+
out.push('-');
|
|
1343
|
+
out.push_str(&hi);
|
|
1344
|
+
i = next;
|
|
1345
|
+
} else {
|
|
1346
|
+
out.push_str(&atom);
|
|
1347
|
+
}
|
|
1348
|
+
}
|
|
1349
|
+
if members == 0 {
|
|
1350
|
+
return None; // `[]` and `[^]` have no counterpart.
|
|
1351
|
+
}
|
|
1352
|
+
out.push(']');
|
|
1353
|
+
Some((out, i))
|
|
1354
|
+
}
|
|
1355
|
+
|
|
1356
|
+
#[cfg(test)]
|
|
1357
|
+
mod tests {
|
|
1358
|
+
use super::*;
|
|
1359
|
+
|
|
1360
|
+
const PATTERNS: &[&str] = &[
|
|
1361
|
+
"",
|
|
1362
|
+
".*",
|
|
1363
|
+
"^.*",
|
|
1364
|
+
".*$",
|
|
1365
|
+
"[\\s\\S]*",
|
|
1366
|
+
"^[\\s\\S]*",
|
|
1367
|
+
"^[\\s\\S]*$",
|
|
1368
|
+
"^.*$",
|
|
1369
|
+
"^[@$_#]",
|
|
1370
|
+
"^[a-zA-Z0-9_\\.\\-\\|@#]*$",
|
|
1371
|
+
"^[a-zA-Z0-9_\\-]*$",
|
|
1372
|
+
".+",
|
|
1373
|
+
"^\\{\\{[^\\W\\.\\-][\\w\\.\\-]*\\}\\}$",
|
|
1374
|
+
"^.{1,256}$",
|
|
1375
|
+
"^[A-Z0-9_\\-\\/]+$",
|
|
1376
|
+
"^[a-zA-Z0-9_\\.\\-]+[\\|]?[a-zA-Z0-9_\\.\\-]+$",
|
|
1377
|
+
"^([a-zA-Z_$][a-zA-Z0-9_$]{0,39}\\.)*([a-zA-Z_$][a-zA-Z0-9_$]{0,39})$",
|
|
1378
|
+
"^x-",
|
|
1379
|
+
"^[1-5](?:[0-9]{2}|XX)$",
|
|
1380
|
+
"(base64key|awskms)://(.*)",
|
|
1381
|
+
"^[A-F0-9]{1,32}$",
|
|
1382
|
+
"^[a-z][a-z0-9]{0,29}$",
|
|
1383
|
+
"^([t|T][o|O][p|P])|([c|C][e|E][n|N][t|T][e|E][r|R])$",
|
|
1384
|
+
"^[\\w\\*]{0,60}$",
|
|
1385
|
+
"^(?:@[0-9a-z-_.]+\\/)?[a-z][0-9a-z-_.]*$",
|
|
1386
|
+
"^[a-z][a-z0-9_]+$",
|
|
1387
|
+
"^\\d+[:-]\\d+$",
|
|
1388
|
+
"^#[0-9a-fA-F]{6}$",
|
|
1389
|
+
"^[^:]+:[^:]+$",
|
|
1390
|
+
"^(?=[^!*,;{}[\\]~\\n]+$)(?=(.*\\w)).+$",
|
|
1391
|
+
"^.*\\.(?:txt|trie)(?:\\.gz)?$",
|
|
1392
|
+
"^([-\\w_\\s]+)(,[-\\w_\\s]+)*$",
|
|
1393
|
+
"^(!?[-\\w_\\s]+)|(\\*)$",
|
|
1394
|
+
"^[0-9]+(ns|ms|us|µs|s|m|h)$",
|
|
1395
|
+
"^\\/[^\\*\\?\\&\\%]*(\\/\\*)?$",
|
|
1396
|
+
"^[^- @#$%^&()!]+$",
|
|
1397
|
+
"^((\\.(?!\\.)\\/)?\\w+\\/?)+$",
|
|
1398
|
+
"^[0-9]{1,}.[0-9]{1,}.[0-9]{1,}$",
|
|
1399
|
+
"\\{.*\\}",
|
|
1400
|
+
"^[a-z]{1,2}$",
|
|
1401
|
+
"^abc$",
|
|
1402
|
+
"^\\/",
|
|
1403
|
+
"^es$",
|
|
1404
|
+
"^(0|[1-9]\\d*)\\.(0|[1-9]\\d*)\\.(0|[1-9]\\d*)(?:-((?:0|[1-9]\\d*|\\d*[a-zA-Z-][0-9a-zA-Z-]*)(?:\\.(?:0|[1-9]\\d*|\\d*[a-zA-Z-][0-9a-zA-Z-]*))*))?(?:\\+([0-9a-zA-Z-]+(?:\\.[0-9a-zA-Z-]+)*))?$",
|
|
1405
|
+
"^[Ee][Ss]5|[Ee][Ss]6|[Ee][Ss]7$",
|
|
1406
|
+
"^[a-z]*a$",
|
|
1407
|
+
"^a*a",
|
|
1408
|
+
"^a+b?a",
|
|
1409
|
+
"^a+b?c*a$",
|
|
1410
|
+
"^[a-z]+-?[a-z]+$",
|
|
1411
|
+
"\\bfoo",
|
|
1412
|
+
"^\\p{L}+$",
|
|
1413
|
+
"\\u00e9",
|
|
1414
|
+
"\\ud83d\\ude00",
|
|
1415
|
+
"[^\\d]x",
|
|
1416
|
+
"^\\S+$",
|
|
1417
|
+
"a{2}b{1,}c{0,3}",
|
|
1418
|
+
"(?<name>ab)+",
|
|
1419
|
+
"^[\\b]",
|
|
1420
|
+
"\\x41",
|
|
1421
|
+
".",
|
|
1422
|
+
"^.",
|
|
1423
|
+
"^.+",
|
|
1424
|
+
"(.*)",
|
|
1425
|
+
"^(.*)",
|
|
1426
|
+
"^.+$",
|
|
1427
|
+
"^.{1,3}$",
|
|
1428
|
+
"^.{2}$",
|
|
1429
|
+
"^.{2,}$",
|
|
1430
|
+
"^(ab|cd)$",
|
|
1431
|
+
"^(?:es|ES|x-|a\\.b)$",
|
|
1432
|
+
"^(a|b|c|d|e|f|g|h|i|j|z)$",
|
|
1433
|
+
"^ab|cd$",
|
|
1434
|
+
"^x-|es|ms$",
|
|
1435
|
+
"a|b",
|
|
1436
|
+
"^a\\$|b",
|
|
1437
|
+
"\\Bs",
|
|
1438
|
+
"a\\b",
|
|
1439
|
+
"^\\p{Lu}",
|
|
1440
|
+
"[\\p{L}\\d]+$",
|
|
1441
|
+
"^\\P{L}+$",
|
|
1442
|
+
"^(?=[^a-c\\n]+$)(?=(.*\\w)).+$",
|
|
1443
|
+
"^\\-a",
|
|
1444
|
+
"[{}[\\]]",
|
|
1445
|
+
"(.+)",
|
|
1446
|
+
"^(.+)$",
|
|
1447
|
+
"^(.*)$",
|
|
1448
|
+
"^a(bc)?$",
|
|
1449
|
+
"^(a|b)c|d(e|f)$",
|
|
1450
|
+
"^([a|A][u|U][t|T][o|O])|([n|N][o|O][n|N][e|E])$",
|
|
1451
|
+
"^[Ee][Ss]2015(\\.([Cc][Oo][Rr][Ee]|[Pp][Rr][Oo][Xx][Yy]))?$",
|
|
1452
|
+
"^[Ee][Ss]([356]|20(1[567]|2[02])|[Nn][Ee][Xx][Tt])$",
|
|
1453
|
+
"^([a-z]+|x)-$",
|
|
1454
|
+
"^a|[0-9]{2}$",
|
|
1455
|
+
"^(a|b)*$",
|
|
1456
|
+
"^(?=a)a|b$",
|
|
1457
|
+
"^/.*",
|
|
1458
|
+
"^a.*",
|
|
1459
|
+
"^a\\\\.*",
|
|
1460
|
+
"(^([0-9]+)\\.([0-9]+)$)|(^\\{[A-F0-9]{2}(-[A-F0-9]{1}){2}\\}$)",
|
|
1461
|
+
"^[0-9]{1,}.[0-9]{1,}$",
|
|
1462
|
+
"^3\\.1\\.\\d+(-.+)?$",
|
|
1463
|
+
"^([A-Za-z_][-A-Za-z0-9_.:]*)$",
|
|
1464
|
+
"^a.c$",
|
|
1465
|
+
"^[a-z].$",
|
|
1466
|
+
"x.y|^z",
|
|
1467
|
+
"^es|ms|x-$",
|
|
1468
|
+
"^(ab){2}$",
|
|
1469
|
+
"^(a|b){2}c$",
|
|
1470
|
+
"^([a-zA-Z0-9]{2,3})(-[a-zA-Z0-9]{1,6})*$",
|
|
1471
|
+
"^([a-z][a-z0-9]{0,3})(\\.[a-z][a-z0-9]{0,3})*$",
|
|
1472
|
+
"^([a-z_$][a-z0-9_$]{0,3}\\.)*([a-zA-Z_$][a-zA-Z0-9_$]{0,3})$",
|
|
1473
|
+
"^([a-z]+)(,[a-z]+)+$",
|
|
1474
|
+
"^(a,)*b$",
|
|
1475
|
+
"^(ab,)+a$",
|
|
1476
|
+
"^a(,a)*$",
|
|
1477
|
+
"^[a-z]*(-[a-z]*)*$",
|
|
1478
|
+
"^(a-)*a-b$",
|
|
1479
|
+
"^(é,)*a$",
|
|
1480
|
+
"^(?=!+[^!*,;{}[\\]~\\n]+$)(?=(.*\\w)).+$",
|
|
1481
|
+
"^(?=!+[^a]+$)(?=(.*\\w)).+$",
|
|
1482
|
+
// Valid only without the `u` flag (identity escapes), so matching code units.
|
|
1483
|
+
"^\\/[^\\*\\?\\&\\%]*(\\/\\*)?$",
|
|
1484
|
+
"^[\\&\\@\\_]+$",
|
|
1485
|
+
"a\\&.b",
|
|
1486
|
+
"^.{2}$",
|
|
1487
|
+
"^[^\\%]{1,3}$",
|
|
1488
|
+
"\\uD83D\\uDE00|\\&",
|
|
1489
|
+
];
|
|
1490
|
+
|
|
1491
|
+
/// Every matcher agrees with regress on strings over an alphabet that exercises classes, anchors and non-ASCII.
|
|
1492
|
+
#[test]
|
|
1493
|
+
fn matchers_agree_with_regress() {
|
|
1494
|
+
let alphabet = [
|
|
1495
|
+
"a", "b", "z", "A", "X", "Z", "0", "1", "5", "9", "_", "-", ".", ":", "/", "@", "#", "$", "*", "!", "{",
|
|
1496
|
+
"}", "|", " ", "\n", "\u{a0}", "é", "µ", "😀", "\u{2028}", "x-", "es", "ES", "ms", "txt", "Au", "to", "No",
|
|
1497
|
+
"ne", "2015", "Co", "re", "20", "15", "22", "2", ",", "a,", "ab,", "a-",
|
|
1498
|
+
];
|
|
1499
|
+
let mut seed: u64 = 0x2545_f491_4f6c_dd1d;
|
|
1500
|
+
let mut next = || {
|
|
1501
|
+
seed ^= seed << 13;
|
|
1502
|
+
seed ^= seed >> 7;
|
|
1503
|
+
seed ^= seed << 17;
|
|
1504
|
+
seed
|
|
1505
|
+
};
|
|
1506
|
+
for &p in PATTERNS {
|
|
1507
|
+
let (reference, _) = compile_regress(p).unwrap();
|
|
1508
|
+
let compiled = compile(p).unwrap();
|
|
1509
|
+
for _ in 0..4000 {
|
|
1510
|
+
let len = next() % 9;
|
|
1511
|
+
let s: String = (0..len).map(|_| alphabet[(next() % alphabet.len() as u64) as usize]).collect();
|
|
1512
|
+
assert_eq!(compiled.is_match(&s), reference.find(&s).is_some(), "{p:?} on {s:?}");
|
|
1513
|
+
}
|
|
1514
|
+
}
|
|
1515
|
+
}
|
|
1516
|
+
|
|
1517
|
+
#[test]
|
|
1518
|
+
fn simple_patterns_take_the_fast_matchers() {
|
|
1519
|
+
for p in ["^[@$_#]", "^[a-zA-Z0-9_\\-]*$", "^#[0-9a-fA-F]{6}$", "^[a-z][a-z0-9_]+$", "^\\d{4}-\\d{2}-\\d{2}$"] {
|
|
1520
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::Sequence(_)), "{p}");
|
|
1521
|
+
}
|
|
1522
|
+
for p in ["^x-", "^\\/", "^abc$"] {
|
|
1523
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::Literal { .. }), "{p}");
|
|
1524
|
+
}
|
|
1525
|
+
for p in ["^[a-z]*a$", "(base64key|awskms)://(.*)"] {
|
|
1526
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::Regex(_)), "{p}");
|
|
1527
|
+
}
|
|
1528
|
+
for p in [".+", "^.+", "(.+)"] {
|
|
1529
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::HasContent { .. }), "{p}");
|
|
1530
|
+
}
|
|
1531
|
+
for p in ["^.{1,256}$", "^.+$", "^.*$", "^(.*)$", "^(.+)$"] {
|
|
1532
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::Line { .. }), "{p}");
|
|
1533
|
+
}
|
|
1534
|
+
for p in ["^(ab|cd)$", "^(?:es|ES|x-)$"] {
|
|
1535
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::Literals(_)), "{p}");
|
|
1536
|
+
}
|
|
1537
|
+
for p in [
|
|
1538
|
+
"^ab|cd$",
|
|
1539
|
+
"a|b",
|
|
1540
|
+
"^([a|A][u|U][t|T][o|O])|([n|N][o|O][n|N][e|E])$",
|
|
1541
|
+
"^[Ee][Ss]2015(\\.([Cc][Oo][Rr][Ee]|[Pp][Rr][Oo][Xx][Yy]))?$",
|
|
1542
|
+
"^[1-5](?:[0-9]{2}|XX)$",
|
|
1543
|
+
"(^([0-9]+)\\.([0-9]+)$)|(^\\{[A-F0-9]{8}(-[A-F0-9]{4}){3}-[A-F0-9]{12}\\}$)",
|
|
1544
|
+
"^([t|T][o|O][p|P])|([c|C][e|E][n|N][t|T][e|E][r|R])|([b|B][o|O][t|T][t|T][o|O][m|M])$",
|
|
1545
|
+
"^([A-Za-z_][-A-Za-z0-9_.:]*)$",
|
|
1546
|
+
] {
|
|
1547
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::Alternatives(_)), "{p}");
|
|
1548
|
+
}
|
|
1549
|
+
for p in [
|
|
1550
|
+
"^([a-zA-Z0-9]{2,3})(-[a-zA-Z0-9]{1,6})*$",
|
|
1551
|
+
"^([a-zA-Z_$][a-zA-Z0-9_$]{0,39}\\.)*([a-zA-Z_$][a-zA-Z0-9_$]{0,39})$",
|
|
1552
|
+
"^([a-z_$][a-z0-9_$]{0,39}\\.)*([a-zA-Z_$][a-zA-Z0-9_$]{0,39})$",
|
|
1553
|
+
] {
|
|
1554
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::SeparatedList(_)), "{p}");
|
|
1555
|
+
}
|
|
1556
|
+
assert!(matches!(
|
|
1557
|
+
compile("^(?=[^!*,;{}[\\]~\\n]+$)(?=(.*\\w)).+$").unwrap().matcher,
|
|
1558
|
+
Matcher::ExcludedClassWithWord { .. }
|
|
1559
|
+
));
|
|
1560
|
+
for p in ["\\bfoo", "^\\p{L}+$"] {
|
|
1561
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::Regex(_)), "{p}");
|
|
1562
|
+
}
|
|
1563
|
+
for p in ["^\\/[^\\*\\?\\&\\%]*(\\/\\*)?$", "a\\&.b", "^\\-a"] {
|
|
1564
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::RegexBmp(..)), "{p}");
|
|
1565
|
+
}
|
|
1566
|
+
for p in ["^((\\.(?!\\.)\\/)?\\w+\\/?)+$", "^\\1(a)", "^\\p{L}\\&"] {
|
|
1567
|
+
assert!(matches!(compile(p).unwrap().matcher, Matcher::Regress(_)), "{p}");
|
|
1568
|
+
}
|
|
1569
|
+
}
|
|
1570
|
+
}
|
|
1571
|
+
|
|
1572
|
+
#[cfg(test)]
|
|
1573
|
+
mod corpus_patterns {
|
|
1574
|
+
/// Lists which matcher each pattern in `CORVUS_PATTERNS` (corpus<TAB>pattern lines) gets.
|
|
1575
|
+
#[test]
|
|
1576
|
+
#[ignore]
|
|
1577
|
+
fn classify_corpus_patterns() {
|
|
1578
|
+
let Ok(path) = std::env::var("CORVUS_PATTERNS") else { return };
|
|
1579
|
+
for line in std::fs::read_to_string(path).unwrap().lines() {
|
|
1580
|
+
let (corpus, p) = line.split_once('\t').unwrap();
|
|
1581
|
+
let kind = super::compile(p)
|
|
1582
|
+
.map(|c| match &c.matcher {
|
|
1583
|
+
super::Matcher::Regex(_) => "regex",
|
|
1584
|
+
super::Matcher::Regress(_) => "regress",
|
|
1585
|
+
_ => "fast",
|
|
1586
|
+
})
|
|
1587
|
+
.unwrap_or("invalid");
|
|
1588
|
+
println!("{kind}\t{corpus}\t{p}");
|
|
1589
|
+
}
|
|
1590
|
+
}
|
|
1591
|
+
}
|