parse-stack-next 5.8.1 → 5.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -18,39 +18,659 @@ module Parse
18
18
  # Maximum allowed length for regex patterns
19
19
  MAX_PATTERN_LENGTH = 500
20
20
 
21
- # Patterns that can cause exponential backtracking in PCRE
22
- DANGEROUS_PATTERNS = [
23
- /\(\?\=|\(\?\!|\(\?\<[!=]/, # Lookahead/lookbehind assertions
24
- /\{(\d{3,}|\d+,\d{3,})\}/, # Large repetition counts {1000} or {1,1000}
25
- /(\.\*|\.\+)\s*(\.\*|\.\+)/, # Consecutive .* or .+ patterns
26
- /\([^)]*(\+|\*)[^)]*\)\s*(\+|\*)/, # Nested quantifiers like (a+)+
27
- /\(\?[^)]*\([^)]*(\+|\*)[^)]*\)[^)]*(\+|\*)\)/, # More complex nested quantifiers
28
- ].freeze
21
+ # `$options` flags accepted on a caller-supplied `$regex`:
22
+ # case-insensitive, multiline, dot-all, and Unicode (`u`, emitted by the
23
+ # unicode form of the regex constraints). The extended flag `x` is
24
+ # refused: it turns whitespace and `#` into comments, which can hide a
25
+ # quantifier from any check that reads the pattern text. Parse Server
26
+ # accepts `x`; this SDK refuses it on input it did not build.
27
+ ALLOWED_OPTIONS = "imsu"
28
+
29
+ # A literal text optionally anchored or wrapped in `.*`: what
30
+ # `starts_with`, `ends_with`, and `contains` build from escaped input.
31
+ # Every regex metacharacter in the body is backslash-escaped, so the
32
+ # pattern cannot backtrack catastrophically.
33
+ LITERAL_BODY = /\A(?:\\.|[^\\.^$|?*+()\[\]{}])*\z/m
34
+
35
+ # Largest repeat count PCRE accepts in `{n}` / `{n,m}`.
36
+ MAX_REPEAT_COUNT = 65_535
37
+
38
+ # Most unbounded or wide-range quantified atoms (`*`, `+`, `{n,}`, or
39
+ # `{n,m}` spanning more than {WIDE_RANGE}) allowed in one sequence
40
+ # before a literal character none of them can match. Adjacent
41
+ # overlapping runs such as `\d+\d+\d+\d+` or `.*a.*a.*a.*a` backtrack
42
+ # polynomially, which at a few hundred characters per document costs as
43
+ # much as the nested shapes the checker refuses.
44
+ MAX_WIDE_RUN = 3
45
+
46
+ # A `{n,m}` range wider than this counts as wide.
47
+ WIDE_RANGE = 16
48
+
49
+ # Raised by {Parser} for a pattern the checker refuses or cannot read.
50
+ # @!visibility private
51
+ class Refused < StandardError; end
52
+
53
+ # @!visibility private
54
+ # A small reader for the PCRE pattern language, enough to find the
55
+ # shapes that backtrack catastrophically. It builds a tree of
56
+ # alternations, sequences, and quantified items, and refuses anything
57
+ # it does not understand (fail closed).
58
+ #
59
+ # Node shapes:
60
+ # [:alt, [seq, ...]] alternation (one seq per branch)
61
+ # [:seq, [item, ...]] concatenation
62
+ # item = { atom:, quant: } quant is nil or { min:, max: } (max nil = unbounded)
63
+ # atom = [:char] | [:dot] | [:anchor] | [:backref] | [:group, alt]
64
+ class Parser
65
+ def initialize(source)
66
+ @src = source
67
+ @pos = 0
68
+ end
69
+
70
+ # @return [Array] the parsed tree.
71
+ def parse
72
+ tree = parse_alt
73
+ refuse("unbalanced ')'") if @pos < @src.length
74
+ tree
75
+ end
76
+
77
+ private
78
+
79
+ def refuse(reason)
80
+ raise Refused, reason
81
+ end
82
+
83
+ def peek(offset = 0)
84
+ @src[@pos + offset]
85
+ end
86
+
87
+ def parse_alt
88
+ branches = [parse_seq]
89
+ while peek == "|"
90
+ @pos += 1
91
+ branches << parse_seq
92
+ end
93
+ [:alt, branches]
94
+ end
95
+
96
+ def parse_seq
97
+ items = []
98
+ while @pos < @src.length && peek != "|" && peek != ")"
99
+ start = @pos
100
+ atom = parse_atom
101
+ next if atom.nil?
102
+ text = @src[start...@pos]
103
+ quant = parse_quant
104
+ if quant && (atom.first == :anchor)
105
+ refuse("quantifier on an anchor or assertion")
106
+ end
107
+ items << { atom: atom, quant: quant, text: text }
108
+ end
109
+ [:seq, items]
110
+ end
111
+
112
+ def parse_atom
113
+ c = peek
114
+ case c
115
+ when "("
116
+ parse_group
117
+ when "["
118
+ parse_class
119
+ [:char]
120
+ when "\\"
121
+ parse_escape
122
+ when "."
123
+ @pos += 1
124
+ [:dot]
125
+ when "^", "$"
126
+ @pos += 1
127
+ [:anchor]
128
+ when "*", "+", "?"
129
+ refuse("quantifier with nothing to repeat")
130
+ when "{"
131
+ if quantifier_at?(@pos)
132
+ refuse("quantifier with nothing to repeat")
133
+ end
134
+ # `{ 2,}` is literal text on the PCRE2 that MongoDB bundles today,
135
+ # but newer PCRE2 releases read it as a repeat count. Refuse it so
136
+ # an upgrade cannot turn it into an unchecked quantifier.
137
+ spaced = @src[@pos..].match(/\A\{[ \d,]*\}/)
138
+ if spaced && spaced[0].include?(" ") && spaced[0].match?(/\d/)
139
+ refuse("whitespace inside a repeat count")
140
+ end
141
+ @pos += 1
142
+ [:char]
143
+ else
144
+ @pos += 1
145
+ [:char]
146
+ end
147
+ end
148
+
149
+ def parse_escape
150
+ @pos += 1
151
+ c = peek
152
+ refuse("trailing backslash") if c.nil?
153
+ @pos += 1
154
+ case c
155
+ when "Q"
156
+ close = @src.index("\\E", @pos)
157
+ @pos = close ? close + 2 : @src.length
158
+ [:char]
159
+ when "1".."9"
160
+ @pos += 1 while peek && peek.match?(/\d/)
161
+ [:backref]
162
+ when "k"
163
+ skip_braced_name
164
+ [:backref]
165
+ when "g"
166
+ # `\g<n>`, `\g'n'`, and `\g<name>` call a group again (its
167
+ # quantifiers included), like `(?1)`; `\g{n}` and `\gN` are
168
+ # backreferences.
169
+ refuse("subroutine call \\g<...> is not allowed") if peek == "<" || peek == "'"
170
+ skip_braced_name
171
+ [:backref]
172
+ when "p", "P", "x", "o", "N"
173
+ skip_braced_name if peek == "{"
174
+ [:char]
175
+ when "c"
176
+ @pos += 1
177
+ [:char]
178
+ when "b", "B", "A", "z", "Z", "G", "K"
179
+ [:anchor]
180
+ else
181
+ [:char]
182
+ end
183
+ end
184
+
185
+ def skip_braced_name
186
+ open = peek
187
+ close = { "{" => "}", "<" => ">", "'" => "'" }[open]
188
+ if close
189
+ finish = @src.index(close, @pos + 1)
190
+ refuse("unterminated escape") if finish.nil?
191
+ @pos = finish + 1
192
+ else
193
+ @pos += 1 while peek && peek.match?(/[-+\d]/)
194
+ end
195
+ end
196
+
197
+ def parse_class
198
+ @pos += 1
199
+ @pos += 1 if peek == "^"
200
+ if peek == "]"
201
+ @pos += 1
202
+ end
203
+ loop do
204
+ c = peek
205
+ refuse("unterminated character class") if c.nil?
206
+ if c == "\\"
207
+ @pos += 2
208
+ elsif c == "[" && %w[: . =].include?(peek(1))
209
+ # PCRE2 reads `[:name:]` (and `[.x.]`, `[=x=]`) only when it is
210
+ # complete right here; otherwise `[` is a literal and the next
211
+ # `]` closes the class.
212
+ posix = @src[@pos..].match(/\A\[([:.=])\^?[a-zA-Z]+\1\]/)
213
+ @pos += posix ? posix[0].length : 1
214
+ elsif c == "]"
215
+ @pos += 1
216
+ break
217
+ else
218
+ @pos += 1
219
+ end
220
+ end
221
+ end
222
+
223
+ def parse_group
224
+ @pos += 1
225
+ kind = :capture
226
+ if peek == "?"
227
+ @pos += 1
228
+ c = peek
229
+ case c
230
+ when ":", ">", "|"
231
+ @pos += 1
232
+ kind = :group
233
+ when "=", "!"
234
+ @pos += 1
235
+ kind = :lookaround
236
+ when "<"
237
+ if peek(1) == "=" || peek(1) == "!"
238
+ @pos += 2
239
+ kind = :lookaround
240
+ else
241
+ skip_group_name(">")
242
+ end
243
+ when "P"
244
+ @pos += 1
245
+ refuse("unsupported group construct") unless peek == "<"
246
+ skip_group_name(">")
247
+ when "'"
248
+ skip_group_name("'")
249
+ when "#"
250
+ refuse("inline comment (?#...)")
251
+ else
252
+ return parse_inline_flags
253
+ end
254
+ end
255
+ body = parse_alt
256
+ refuse("unbalanced '('") unless peek == ")"
257
+ @pos += 1
258
+ [:group, body, kind]
259
+ end
260
+
261
+ def skip_group_name(close)
262
+ finish = @src.index(close, @pos + 1)
263
+ refuse("unterminated group name") if finish.nil?
264
+ @pos = finish + 1
265
+ end
266
+
267
+ # `(?flags)` or `(?flags:...)`, where flags are `on-off`. The
268
+ # extended flag turned on is refused; anything that is not a flag
269
+ # group (recursion, conditionals, callouts) is refused too.
270
+ def parse_inline_flags
271
+ start = @pos
272
+ @pos += 1 while peek && peek.match?(/[a-zA-Z^-]/)
273
+ flags = @src[start...@pos]
274
+ refuse("unsupported group construct (?#{peek})") if flags.empty?
275
+ on = flags.sub(/\A\^/, "").split("-", 2).first.to_s
276
+ refuse("extended mode (?x) is not allowed") if on.include?("x")
277
+ unless flags.match?(/\A\^?[imsnUJ]*(?:-[imsnUJ]*)?\z/) || flags.match?(/\A\^?[imsxnUJ]*-[imsxnUJ]*\z/)
278
+ refuse("unsupported inline flags (?#{flags})")
279
+ end
280
+ if peek == ")"
281
+ @pos += 1
282
+ return nil
283
+ end
284
+ refuse("unsupported group construct") unless peek == ":"
285
+ @pos += 1
286
+ body = parse_alt
287
+ refuse("unbalanced '('") unless peek == ")"
288
+ @pos += 1
289
+ [:group, body, :group]
290
+ end
291
+
292
+ def quantifier_at?(idx)
293
+ m = scan_brace_quantifier(idx)
294
+ !m.nil? && !(m[0].empty? && m[2].empty?)
295
+ end
296
+
297
+ # Read a `{n}`, `{n,}`, `{,m}`, or `{n,m}` count starting at `idx` with a
298
+ # linear scan. Returns `[min_digits, comma, max_digits, length]`, or nil
299
+ # when the text there is not a complete count.
300
+ def scan_brace_quantifier(idx)
301
+ return nil unless @src[idx] == "{"
302
+ i = idx + 1
303
+ j = i
304
+ j += 1 while j < @src.length && @src[j].match?(/\d/)
305
+ min_digits = @src[i...j]
306
+ comma = ""
307
+ if @src[j] == ","
308
+ comma = ","
309
+ j += 1
310
+ end
311
+ k = j
312
+ k += 1 while k < @src.length && @src[k].match?(/\d/)
313
+ max_digits = @src[j...k]
314
+ return nil unless @src[k] == "}"
315
+ [min_digits, comma, max_digits, k - idx + 1]
316
+ end
317
+
318
+ def parse_quant
319
+ c = peek
320
+ quant = case c
321
+ when "*" then @pos += 1; { min: 0, max: nil }
322
+ when "+" then @pos += 1; { min: 1, max: nil }
323
+ when "?" then @pos += 1; { min: 0, max: 1 }
324
+ when "{"
325
+ m = scan_brace_quantifier(@pos)
326
+ return nil if m.nil? || (m[0].empty? && m[2].empty?)
327
+ @pos += m[3]
328
+ min = m[0].empty? ? 0 : m[0].to_i
329
+ max = if m[1].empty? then min
330
+ elsif m[2].empty? then nil
331
+ else m[2].to_i
332
+ end
333
+ if min > MAX_REPEAT_COUNT || (max && max > MAX_REPEAT_COUNT)
334
+ refuse("repeat count above #{MAX_REPEAT_COUNT}")
335
+ end
336
+ refuse("repeat range out of order") if max && max < min
337
+ { min: min, max: max }
338
+ end
339
+ return nil if quant.nil?
340
+ @pos += 1 if peek == "?" || peek == "+"
341
+ refuse("stacked quantifiers") if %w[* + ?].include?(peek) || (peek == "{" && quantifier_at?(@pos))
342
+ quant
343
+ end
344
+ end
29
345
 
30
346
  class << self
31
347
  # Validates a regex pattern for potential ReDoS vulnerabilities.
348
+ #
349
+ # Escaped literal text (what `starts_with`, `ends_with`, and
350
+ # `contains` build) always passes; its length cap is
351
+ # `2 * max_length + 4` because escaping can double it. Any other
352
+ # pattern is parsed and refused when it contains a shape that
353
+ # backtracks catastrophically on PCRE:
354
+ #
355
+ # * a group repeated more than once (`+`, `*`, `{n,}`, `{n}` or
356
+ # `{n,m}` with a top above 1) whose body holds a quantifier, an
357
+ # alternation, or a backreference: `(a+)+`, `(a|aa)+`, `((a+))+`,
358
+ # `(a+){2,}`, `(a+){20}`;
359
+ # A repeated group stays allowed when each repeat cannot split its
360
+ # input more than one way: an alternation of fixed literals with
361
+ # distinct first characters (`(foo|bar)+`), or a literal separator
362
+ # next to one quantified atom that cannot match it
363
+ # (`(-[a-z0-9]+)*`, `(\.[\w-]+)+`, `([\w-]+\.)+`);
364
+ # * two adjacent unbounded `.*` / `.+` with more pattern after them;
365
+ # * more than three unbounded or wide quantified atoms in a row with
366
+ # no literal between them that they cannot match
367
+ # (`\d+\d+\d+\d+`, `.*a.*a.*a.*a`); groups without a quantifier are
368
+ # read as part of the surrounding sequence;
369
+ # * a subroutine call (`\g<1>`), whitespace inside a repeat count
370
+ # (`{ 2,}`), which newer PCRE2 reads as a quantifier;
371
+ # * an inline comment `(?#...)`, extended mode (`(?x)` or a Regexp
372
+ # with `Regexp::EXTENDED`), recursion, conditionals, or anything
373
+ # else the reader does not recognize, and unbalanced patterns.
374
+ #
375
+ # Repeats of a single atom (`.{1,255}`, `\d{1,100}`, `x{1000}`, `a+`)
376
+ # and lookarounds without such a shape inside (`^(?!test)`) pass.
32
377
  # @param pattern [String, Regexp] the pattern to validate
33
378
  # @param max_length [Integer] maximum allowed pattern length
34
379
  # @raise [ArgumentError] if the pattern is potentially dangerous
35
380
  # @return [String] the validated pattern string
36
381
  def validate!(pattern, max_length: MAX_PATTERN_LENGTH)
382
+ if pattern.is_a?(Regexp) && (pattern.options & Regexp::EXTENDED) != 0
383
+ raise ArgumentError, "Regex pattern uses extended mode (the x flag), which is not allowed: " \
384
+ "it can hide quantifiers in comments. Pattern: #{pattern.source.inspect}"
385
+ end
37
386
  pattern_str = pattern.is_a?(Regexp) ? pattern.source : pattern.to_s
387
+ literal = literal_pattern?(pattern_str)
388
+ cap = literal ? (2 * max_length) + 4 : max_length
38
389
 
39
- if pattern_str.length > max_length
40
- raise ArgumentError, "Regex pattern too long (#{pattern_str.length} chars, max #{max_length}). " \
390
+ if pattern_str.length > cap
391
+ raise ArgumentError, "Regex pattern too long (#{pattern_str.length} chars, max #{cap}). " \
41
392
  "Long patterns can cause performance issues."
42
393
  end
43
394
 
44
- DANGEROUS_PATTERNS.each do |dangerous|
45
- if pattern_str.match?(dangerous)
46
- raise ArgumentError, "Regex pattern contains potentially dangerous constructs that could cause " \
47
- "ReDoS (Regular Expression Denial of Service). Pattern: #{pattern_str.inspect}"
395
+ return pattern_str if literal
396
+
397
+ reason = begin
398
+ check_tree(Parser.new(pattern_str).parse)
399
+ rescue Refused => e
400
+ e.message
48
401
  end
402
+ if reason
403
+ raise ArgumentError, "Regex pattern contains potentially dangerous constructs that could cause " \
404
+ "ReDoS (Regular Expression Denial of Service): #{reason}. Pattern: #{pattern_str.inspect}"
49
405
  end
50
406
 
51
407
  pattern_str
52
408
  end
53
409
 
410
+ # Whether a pattern is escaped literal text, optionally anchored
411
+ # (`^text`, `text$`) or wrapped in `.*` (`.*text.*`). A pattern that is
412
+ # only `.*` wrappers (`.*.*`) is not literal text.
413
+ # @param pattern_str [String]
414
+ # @return [Boolean]
415
+ def literal_pattern?(pattern_str)
416
+ body = pattern_str.dup
417
+ dot_prefix = false
418
+ if body.start_with?("^")
419
+ body = body[1..]
420
+ elsif body.start_with?(".*")
421
+ body = body[2..]
422
+ dot_prefix = true
423
+ end
424
+ dot_suffix = false
425
+ if body.end_with?(".*") && !body.end_with?("\\.*")
426
+ body = body[0..-3]
427
+ dot_suffix = true
428
+ elsif body.end_with?("$") && !body.end_with?("\\$")
429
+ body = body[0..-2]
430
+ end
431
+ return false if dot_prefix && dot_suffix && body.empty?
432
+ LITERAL_BODY.match?(body)
433
+ end
434
+
435
+ # Validates `$options` flags against {ALLOWED_OPTIONS}.
436
+ # @param options [Object]
437
+ # @raise [ArgumentError] on a non-String or an unknown flag.
438
+ def validate_options!(options)
439
+ unless options.is_a?(String)
440
+ raise ArgumentError, "Regex $options must be a String (got #{options.class})."
441
+ end
442
+ bad = options.chars.uniq.reject { |c| ALLOWED_OPTIONS.include?(c) }
443
+ unless bad.empty?
444
+ raise ArgumentError, "Regex $options contains unsupported flags #{bad.join.inspect}. " \
445
+ "Allowed: #{ALLOWED_OPTIONS.chars.join(", ")}."
446
+ end
447
+ options
448
+ end
449
+
450
+ # Validates every `$regex` (and its `$options`) inside a compiled where
451
+ # clause, at any depth: field values, `$not` / `$elemMatch` wrappers,
452
+ # and `$or` / `$and` / `$nor` branches, plus Regexp and BSON regex
453
+ # values anywhere (equality, `$not`, `$in`, `$nin`, `$all`). SDK
454
+ # routing markers (`__` keys) are skipped. Literal patterns built from
455
+ # escaped input pass.
456
+ # @param node [Object] a compiled where clause or part of one.
457
+ # @raise [ArgumentError] when a pattern or its options are unsafe.
458
+ # @return [void]
459
+ def validate_where!(node)
460
+ case node
461
+ when Hash
462
+ node.each do |key, value|
463
+ key_str = key.to_s
464
+ next if key_str.start_with?("__")
465
+ if key_str == "$regex"
466
+ value.is_a?(String) ? validate!(value) : validate_where!(value)
467
+ elsif key_str == "$options" && (node.key?("$regex") || node.key?(:$regex))
468
+ validate_options!(value)
469
+ else
470
+ validate_where!(value)
471
+ end
472
+ end
473
+ when Array
474
+ node.each { |item| validate_where!(item) }
475
+ when Regexp
476
+ # A Regexp value (equality, `$not`, `$in`) is a regex match too.
477
+ validate!(node)
478
+ else
479
+ if defined?(BSON::Regexp::Raw) && node.is_a?(BSON::Regexp::Raw)
480
+ if node.options.to_s.include?("x")
481
+ raise ArgumentError, "Regex pattern uses extended mode (the x flag), which is not allowed."
482
+ end
483
+ validate!(node.pattern.to_s)
484
+ end
485
+ end
486
+ nil
487
+ end
488
+
489
+ # @!visibility private
490
+ # The first refused shape in a parsed pattern, or nil.
491
+ def check_tree(node)
492
+ case node.first
493
+ when :alt
494
+ node[1].each do |seq|
495
+ reason = check_tree(seq)
496
+ return reason if reason
497
+ end
498
+ when :seq
499
+ items = node[1]
500
+ reason = wide_run_reason(flatten_items(items))
501
+ return reason if reason
502
+ items.each_with_index do |item, idx|
503
+ atom = item[:atom]
504
+ quant = item[:quant]
505
+ if atom.first == :group
506
+ if repeats?(quant) && complex?(atom[1]) && !separated_repeat?(atom[1])
507
+ return "a repeated group contains a quantifier, alternation, or backreference"
508
+ end
509
+ reason = check_tree(atom[1])
510
+ return reason if reason
511
+ end
512
+ nxt = items[idx + 1]
513
+ if unbounded_dot?(item) && nxt && unbounded_dot?(nxt) &&
514
+ !items[(idx + 2)..].all? { |rest| rest[:atom].first == :anchor }
515
+ return "adjacent unbounded .* or .+ followed by more pattern"
516
+ end
517
+ end
518
+ end
519
+ nil
520
+ end
521
+ private :check_tree
522
+
523
+ # @!visibility private
524
+ # The sequence with non-repeated, non-lookaround groups spliced in,
525
+ # so `(?:.*)(?:.*)` reads as `.*.*`.
526
+ def flatten_items(items)
527
+ items.flat_map do |item|
528
+ atom = item[:atom]
529
+ if atom.first == :group && item[:quant].nil? && atom[2] != :lookaround && atom[1][1].length == 1
530
+ flatten_items(atom[1][1].first[1])
531
+ else
532
+ [item]
533
+ end
534
+ end
535
+ end
536
+ private :flatten_items
537
+
538
+ # @!visibility private
539
+ def wide?(item)
540
+ quant = item[:quant]
541
+ !quant.nil? && (quant[:max].nil? || quant[:max] - quant[:min] > WIDE_RANGE)
542
+ end
543
+ private :wide?
544
+
545
+ # @!visibility private
546
+ # A refusal reason when more than {MAX_WIDE_RUN} wide atoms appear
547
+ # without a literal character between them that none of the wide
548
+ # atoms in the sequence can match.
549
+ def wide_run_reason(items)
550
+ wide_items = items.select { |item| wide?(item) }
551
+ return nil if wide_items.length <= MAX_WIDE_RUN
552
+ count = 0
553
+ items.each do |item|
554
+ if wide?(item)
555
+ count += 1
556
+ return "more than #{MAX_WIDE_RUN} unbounded or wide quantifiers in a row" if count > MAX_WIDE_RUN
557
+ elsif (ch = literal_char(item)) && wide_items.none? { |w| atom_matches?(w, ch) }
558
+ count = 0
559
+ end
560
+ end
561
+ nil
562
+ end
563
+ private :wide_run_reason
564
+
565
+ # @!visibility private
566
+ # The character a single unquantified literal atom matches (`a`, `-`,
567
+ # `\.`), or nil.
568
+ def literal_char(item)
569
+ return nil unless item[:quant].nil? && item[:atom] == [:char]
570
+ text = item[:text].to_s
571
+ if text.length == 1 && !text.match?(/[\\.\[\]()|?*+{}^$]/)
572
+ text
573
+ elsif (m = text.match(/\A\\([^A-Za-z0-9])\z/))
574
+ m[1]
575
+ end
576
+ end
577
+ private :literal_char
578
+
579
+ # @!visibility private
580
+ # Whether an atom (ignoring its quantifier) can match `ch` in either
581
+ # case. Anything that cannot be checked counts as a match.
582
+ def atom_matches?(item, ch)
583
+ atom = item[:atom]
584
+ return true unless atom == [:char] || atom == [:dot]
585
+ re = begin
586
+ Regexp.new(atom == [:dot] ? "." : item[:text].to_s)
587
+ rescue RegexpError, ArgumentError
588
+ nil
589
+ end
590
+ return true if re.nil?
591
+ [ch, ch.swapcase, ch.downcase(:fold), ch.upcase].uniq.any? { |c| re.match?(c) }
592
+ end
593
+ private :atom_matches?
594
+
595
+ # @!visibility private
596
+ # A repeated group body that cannot split its input more than one way
597
+ # per repeat, so repeating it is safe:
598
+ #
599
+ # * an alternation of fixed literal strings whose first characters
600
+ # differ, compared case-insensitively because the pattern may run
601
+ # with the `i` option or `(?i)` (`(foo|bar)+`, but not `(a|Aa)+`);
602
+ # * a mandatory literal separator next to one quantified atom that
603
+ # cannot match it (`(-[a-z0-9]+)*`, `(\.[\w-]+)+`, `([\w-]+\.)+`).
604
+ def separated_repeat?(body)
605
+ branches = body[1]
606
+ if branches.length > 1
607
+ firsts = branches.map do |seq|
608
+ items = seq[1]
609
+ return false if items.empty?
610
+ chars = items.map { |item| literal_char(item) }
611
+ return false if chars.any?(&:nil?)
612
+ chars.first.downcase(:fold)
613
+ end
614
+ return firsts.uniq.length == firsts.length
615
+ end
616
+ items = branches.first[1]
617
+ return false unless items.length == 2
618
+ sep, run = items
619
+ sep, run = run, sep if literal_char(sep).nil?
620
+ ch = literal_char(sep)
621
+ return false if ch.nil?
622
+ return false unless run[:quant] && (run[:atom] == [:char] || run[:atom] == [:dot])
623
+ !atom_matches?(run, ch)
624
+ end
625
+ private :separated_repeat?
626
+
627
+ # @!visibility private
628
+ def repeats?(quant)
629
+ !quant.nil? && (quant[:max].nil? || quant[:max] > 1)
630
+ end
631
+ private :repeats?
632
+
633
+ # @!visibility private
634
+ def unbounded_dot?(item)
635
+ item[:atom].first == :dot && item[:quant] && item[:quant][:max].nil?
636
+ end
637
+ private :unbounded_dot?
638
+
639
+ # @!visibility private
640
+ # Whether a quantifier lets the engine choose how many times to match
641
+ # (`?`, `*`, `+`, `{0,n}`, `{n,m}` with m > n, `{n,}`), so each repeat
642
+ # of an enclosing group can split the input another way. A fixed count
643
+ # (`{3}`) gives no choice.
644
+ def branches?(quant)
645
+ !quant.nil? && (quant[:max].nil? || quant[:max] != quant[:min])
646
+ end
647
+ private :branches?
648
+
649
+ # @!visibility private
650
+ # Whether a subtree holds a quantifier that introduces a choice
651
+ # (including an optional `?`, which matches zero or one time), an
652
+ # alternation with more than one branch, or a backreference. Any of
653
+ # these inside a group repeated more than once multiplies the ways a
654
+ # failing match can be retried: `(a?){100}a{100}` backtracks as badly
655
+ # as `(a+)+`.
656
+ def complex?(node)
657
+ case node.first
658
+ when :alt
659
+ return true if node[1].length > 1
660
+ node[1].any? { |seq| complex?(seq) }
661
+ when :seq
662
+ node[1].any? do |item|
663
+ atom = item[:atom]
664
+ branches?(item[:quant]) ||
665
+ atom.first == :backref ||
666
+ (atom.first == :group && complex?(atom[1]))
667
+ end
668
+ else
669
+ false
670
+ end
671
+ end
672
+ private :complex?
673
+
54
674
  # Checks if a pattern is safe without raising an exception.
55
675
  # @param pattern [String, Regexp] the pattern to check
56
676
  # @return [Boolean] true if safe, false if potentially dangerous