parse-stack-next 5.8.1 → 5.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +312 -0
- data/README.md +6 -0
- data/docs/caching.md +40 -5
- data/docs/mcp_guide.md +7 -2
- data/docs/mongodb_direct_guide.md +29 -0
- data/docs/webhooks_guide.md +35 -2
- data/lib/parse/agent/constraint_translator.rb +22 -21
- data/lib/parse/agent/errors.rb +17 -0
- data/lib/parse/agent/mcp_server.rb +47 -6
- data/lib/parse/agent.rb +316 -50
- data/lib/parse/api/batch.rb +70 -2
- data/lib/parse/atlas_search.rb +1 -0
- data/lib/parse/client/batch.rb +231 -12
- data/lib/parse/client/body_builder.rb +100 -15
- data/lib/parse/client/caching.rb +53 -3
- data/lib/parse/client/request.rb +59 -0
- data/lib/parse/client/response.rb +63 -5
- data/lib/parse/client.rb +147 -31
- data/lib/parse/live_query/client.rb +70 -9
- data/lib/parse/model/classes/session.rb +44 -16
- data/lib/parse/model/core/actions.rb +22 -1
- data/lib/parse/model/file.rb +96 -19
- data/lib/parse/mongodb.rb +24 -1
- data/lib/parse/pipeline_security.rb +157 -21
- data/lib/parse/query/constraints.rb +634 -14
- data/lib/parse/query.rb +9 -5
- data/lib/parse/stack/tasks.rb +1 -1
- data/lib/parse/stack/version.rb +1 -1
- data/lib/parse/stack.rb +6 -2
- data/lib/parse/vector_search/hybrid.rb +4 -1
- data/lib/parse/vector_search.rb +1 -0
- data/lib/parse/webhooks/payload.rb +39 -8
- data/lib/parse/webhooks/registration.rb +26 -6
- data/lib/parse/webhooks/replay_protection.rb +19 -9
- data/lib/parse/webhooks.rb +17 -4
- metadata +1 -1
|
@@ -18,39 +18,659 @@ module Parse
|
|
|
18
18
|
# Maximum allowed length for regex patterns
|
|
19
19
|
MAX_PATTERN_LENGTH = 500
|
|
20
20
|
|
|
21
|
-
#
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
21
|
+
# `$options` flags accepted on a caller-supplied `$regex`:
|
|
22
|
+
# case-insensitive, multiline, dot-all, and Unicode (`u`, emitted by the
|
|
23
|
+
# unicode form of the regex constraints). The extended flag `x` is
|
|
24
|
+
# refused: it turns whitespace and `#` into comments, which can hide a
|
|
25
|
+
# quantifier from any check that reads the pattern text. Parse Server
|
|
26
|
+
# accepts `x`; this SDK refuses it on input it did not build.
|
|
27
|
+
ALLOWED_OPTIONS = "imsu"
|
|
28
|
+
|
|
29
|
+
# A literal text optionally anchored or wrapped in `.*`: what
|
|
30
|
+
# `starts_with`, `ends_with`, and `contains` build from escaped input.
|
|
31
|
+
# Every regex metacharacter in the body is backslash-escaped, so the
|
|
32
|
+
# pattern cannot backtrack catastrophically.
|
|
33
|
+
LITERAL_BODY = /\A(?:\\.|[^\\.^$|?*+()\[\]{}])*\z/m
|
|
34
|
+
|
|
35
|
+
# Largest repeat count PCRE accepts in `{n}` / `{n,m}`.
|
|
36
|
+
MAX_REPEAT_COUNT = 65_535
|
|
37
|
+
|
|
38
|
+
# Most unbounded or wide-range quantified atoms (`*`, `+`, `{n,}`, or
|
|
39
|
+
# `{n,m}` spanning more than {WIDE_RANGE}) allowed in one sequence
|
|
40
|
+
# before a literal character none of them can match. Adjacent
|
|
41
|
+
# overlapping runs such as `\d+\d+\d+\d+` or `.*a.*a.*a.*a` backtrack
|
|
42
|
+
# polynomially, which at a few hundred characters per document costs as
|
|
43
|
+
# much as the nested shapes the checker refuses.
|
|
44
|
+
MAX_WIDE_RUN = 3
|
|
45
|
+
|
|
46
|
+
# A `{n,m}` range wider than this counts as wide.
|
|
47
|
+
WIDE_RANGE = 16
|
|
48
|
+
|
|
49
|
+
# Raised by {Parser} for a pattern the checker refuses or cannot read.
|
|
50
|
+
# @!visibility private
|
|
51
|
+
class Refused < StandardError; end
|
|
52
|
+
|
|
53
|
+
# @!visibility private
|
|
54
|
+
# A small reader for the PCRE pattern language, enough to find the
|
|
55
|
+
# shapes that backtrack catastrophically. It builds a tree of
|
|
56
|
+
# alternations, sequences, and quantified items, and refuses anything
|
|
57
|
+
# it does not understand (fail closed).
|
|
58
|
+
#
|
|
59
|
+
# Node shapes:
|
|
60
|
+
# [:alt, [seq, ...]] alternation (one seq per branch)
|
|
61
|
+
# [:seq, [item, ...]] concatenation
|
|
62
|
+
# item = { atom:, quant: } quant is nil or { min:, max: } (max nil = unbounded)
|
|
63
|
+
# atom = [:char] | [:dot] | [:anchor] | [:backref] | [:group, alt]
|
|
64
|
+
class Parser
|
|
65
|
+
def initialize(source)
|
|
66
|
+
@src = source
|
|
67
|
+
@pos = 0
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# @return [Array] the parsed tree.
|
|
71
|
+
def parse
|
|
72
|
+
tree = parse_alt
|
|
73
|
+
refuse("unbalanced ')'") if @pos < @src.length
|
|
74
|
+
tree
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
private
|
|
78
|
+
|
|
79
|
+
def refuse(reason)
|
|
80
|
+
raise Refused, reason
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def peek(offset = 0)
|
|
84
|
+
@src[@pos + offset]
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def parse_alt
|
|
88
|
+
branches = [parse_seq]
|
|
89
|
+
while peek == "|"
|
|
90
|
+
@pos += 1
|
|
91
|
+
branches << parse_seq
|
|
92
|
+
end
|
|
93
|
+
[:alt, branches]
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def parse_seq
|
|
97
|
+
items = []
|
|
98
|
+
while @pos < @src.length && peek != "|" && peek != ")"
|
|
99
|
+
start = @pos
|
|
100
|
+
atom = parse_atom
|
|
101
|
+
next if atom.nil?
|
|
102
|
+
text = @src[start...@pos]
|
|
103
|
+
quant = parse_quant
|
|
104
|
+
if quant && (atom.first == :anchor)
|
|
105
|
+
refuse("quantifier on an anchor or assertion")
|
|
106
|
+
end
|
|
107
|
+
items << { atom: atom, quant: quant, text: text }
|
|
108
|
+
end
|
|
109
|
+
[:seq, items]
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
def parse_atom
|
|
113
|
+
c = peek
|
|
114
|
+
case c
|
|
115
|
+
when "("
|
|
116
|
+
parse_group
|
|
117
|
+
when "["
|
|
118
|
+
parse_class
|
|
119
|
+
[:char]
|
|
120
|
+
when "\\"
|
|
121
|
+
parse_escape
|
|
122
|
+
when "."
|
|
123
|
+
@pos += 1
|
|
124
|
+
[:dot]
|
|
125
|
+
when "^", "$"
|
|
126
|
+
@pos += 1
|
|
127
|
+
[:anchor]
|
|
128
|
+
when "*", "+", "?"
|
|
129
|
+
refuse("quantifier with nothing to repeat")
|
|
130
|
+
when "{"
|
|
131
|
+
if quantifier_at?(@pos)
|
|
132
|
+
refuse("quantifier with nothing to repeat")
|
|
133
|
+
end
|
|
134
|
+
# `{ 2,}` is literal text on the PCRE2 that MongoDB bundles today,
|
|
135
|
+
# but newer PCRE2 releases read it as a repeat count. Refuse it so
|
|
136
|
+
# an upgrade cannot turn it into an unchecked quantifier.
|
|
137
|
+
spaced = @src[@pos..].match(/\A\{[ \d,]*\}/)
|
|
138
|
+
if spaced && spaced[0].include?(" ") && spaced[0].match?(/\d/)
|
|
139
|
+
refuse("whitespace inside a repeat count")
|
|
140
|
+
end
|
|
141
|
+
@pos += 1
|
|
142
|
+
[:char]
|
|
143
|
+
else
|
|
144
|
+
@pos += 1
|
|
145
|
+
[:char]
|
|
146
|
+
end
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
def parse_escape
|
|
150
|
+
@pos += 1
|
|
151
|
+
c = peek
|
|
152
|
+
refuse("trailing backslash") if c.nil?
|
|
153
|
+
@pos += 1
|
|
154
|
+
case c
|
|
155
|
+
when "Q"
|
|
156
|
+
close = @src.index("\\E", @pos)
|
|
157
|
+
@pos = close ? close + 2 : @src.length
|
|
158
|
+
[:char]
|
|
159
|
+
when "1".."9"
|
|
160
|
+
@pos += 1 while peek && peek.match?(/\d/)
|
|
161
|
+
[:backref]
|
|
162
|
+
when "k"
|
|
163
|
+
skip_braced_name
|
|
164
|
+
[:backref]
|
|
165
|
+
when "g"
|
|
166
|
+
# `\g<n>`, `\g'n'`, and `\g<name>` call a group again (its
|
|
167
|
+
# quantifiers included), like `(?1)`; `\g{n}` and `\gN` are
|
|
168
|
+
# backreferences.
|
|
169
|
+
refuse("subroutine call \\g<...> is not allowed") if peek == "<" || peek == "'"
|
|
170
|
+
skip_braced_name
|
|
171
|
+
[:backref]
|
|
172
|
+
when "p", "P", "x", "o", "N"
|
|
173
|
+
skip_braced_name if peek == "{"
|
|
174
|
+
[:char]
|
|
175
|
+
when "c"
|
|
176
|
+
@pos += 1
|
|
177
|
+
[:char]
|
|
178
|
+
when "b", "B", "A", "z", "Z", "G", "K"
|
|
179
|
+
[:anchor]
|
|
180
|
+
else
|
|
181
|
+
[:char]
|
|
182
|
+
end
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
def skip_braced_name
|
|
186
|
+
open = peek
|
|
187
|
+
close = { "{" => "}", "<" => ">", "'" => "'" }[open]
|
|
188
|
+
if close
|
|
189
|
+
finish = @src.index(close, @pos + 1)
|
|
190
|
+
refuse("unterminated escape") if finish.nil?
|
|
191
|
+
@pos = finish + 1
|
|
192
|
+
else
|
|
193
|
+
@pos += 1 while peek && peek.match?(/[-+\d]/)
|
|
194
|
+
end
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
def parse_class
|
|
198
|
+
@pos += 1
|
|
199
|
+
@pos += 1 if peek == "^"
|
|
200
|
+
if peek == "]"
|
|
201
|
+
@pos += 1
|
|
202
|
+
end
|
|
203
|
+
loop do
|
|
204
|
+
c = peek
|
|
205
|
+
refuse("unterminated character class") if c.nil?
|
|
206
|
+
if c == "\\"
|
|
207
|
+
@pos += 2
|
|
208
|
+
elsif c == "[" && %w[: . =].include?(peek(1))
|
|
209
|
+
# PCRE2 reads `[:name:]` (and `[.x.]`, `[=x=]`) only when it is
|
|
210
|
+
# complete right here; otherwise `[` is a literal and the next
|
|
211
|
+
# `]` closes the class.
|
|
212
|
+
posix = @src[@pos..].match(/\A\[([:.=])\^?[a-zA-Z]+\1\]/)
|
|
213
|
+
@pos += posix ? posix[0].length : 1
|
|
214
|
+
elsif c == "]"
|
|
215
|
+
@pos += 1
|
|
216
|
+
break
|
|
217
|
+
else
|
|
218
|
+
@pos += 1
|
|
219
|
+
end
|
|
220
|
+
end
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
def parse_group
|
|
224
|
+
@pos += 1
|
|
225
|
+
kind = :capture
|
|
226
|
+
if peek == "?"
|
|
227
|
+
@pos += 1
|
|
228
|
+
c = peek
|
|
229
|
+
case c
|
|
230
|
+
when ":", ">", "|"
|
|
231
|
+
@pos += 1
|
|
232
|
+
kind = :group
|
|
233
|
+
when "=", "!"
|
|
234
|
+
@pos += 1
|
|
235
|
+
kind = :lookaround
|
|
236
|
+
when "<"
|
|
237
|
+
if peek(1) == "=" || peek(1) == "!"
|
|
238
|
+
@pos += 2
|
|
239
|
+
kind = :lookaround
|
|
240
|
+
else
|
|
241
|
+
skip_group_name(">")
|
|
242
|
+
end
|
|
243
|
+
when "P"
|
|
244
|
+
@pos += 1
|
|
245
|
+
refuse("unsupported group construct") unless peek == "<"
|
|
246
|
+
skip_group_name(">")
|
|
247
|
+
when "'"
|
|
248
|
+
skip_group_name("'")
|
|
249
|
+
when "#"
|
|
250
|
+
refuse("inline comment (?#...)")
|
|
251
|
+
else
|
|
252
|
+
return parse_inline_flags
|
|
253
|
+
end
|
|
254
|
+
end
|
|
255
|
+
body = parse_alt
|
|
256
|
+
refuse("unbalanced '('") unless peek == ")"
|
|
257
|
+
@pos += 1
|
|
258
|
+
[:group, body, kind]
|
|
259
|
+
end
|
|
260
|
+
|
|
261
|
+
def skip_group_name(close)
|
|
262
|
+
finish = @src.index(close, @pos + 1)
|
|
263
|
+
refuse("unterminated group name") if finish.nil?
|
|
264
|
+
@pos = finish + 1
|
|
265
|
+
end
|
|
266
|
+
|
|
267
|
+
# `(?flags)` or `(?flags:...)`, where flags are `on-off`. The
|
|
268
|
+
# extended flag turned on is refused; anything that is not a flag
|
|
269
|
+
# group (recursion, conditionals, callouts) is refused too.
|
|
270
|
+
def parse_inline_flags
|
|
271
|
+
start = @pos
|
|
272
|
+
@pos += 1 while peek && peek.match?(/[a-zA-Z^-]/)
|
|
273
|
+
flags = @src[start...@pos]
|
|
274
|
+
refuse("unsupported group construct (?#{peek})") if flags.empty?
|
|
275
|
+
on = flags.sub(/\A\^/, "").split("-", 2).first.to_s
|
|
276
|
+
refuse("extended mode (?x) is not allowed") if on.include?("x")
|
|
277
|
+
unless flags.match?(/\A\^?[imsnUJ]*(?:-[imsnUJ]*)?\z/) || flags.match?(/\A\^?[imsxnUJ]*-[imsxnUJ]*\z/)
|
|
278
|
+
refuse("unsupported inline flags (?#{flags})")
|
|
279
|
+
end
|
|
280
|
+
if peek == ")"
|
|
281
|
+
@pos += 1
|
|
282
|
+
return nil
|
|
283
|
+
end
|
|
284
|
+
refuse("unsupported group construct") unless peek == ":"
|
|
285
|
+
@pos += 1
|
|
286
|
+
body = parse_alt
|
|
287
|
+
refuse("unbalanced '('") unless peek == ")"
|
|
288
|
+
@pos += 1
|
|
289
|
+
[:group, body, :group]
|
|
290
|
+
end
|
|
291
|
+
|
|
292
|
+
def quantifier_at?(idx)
|
|
293
|
+
m = scan_brace_quantifier(idx)
|
|
294
|
+
!m.nil? && !(m[0].empty? && m[2].empty?)
|
|
295
|
+
end
|
|
296
|
+
|
|
297
|
+
# Read a `{n}`, `{n,}`, `{,m}`, or `{n,m}` count starting at `idx` with a
|
|
298
|
+
# linear scan. Returns `[min_digits, comma, max_digits, length]`, or nil
|
|
299
|
+
# when the text there is not a complete count.
|
|
300
|
+
def scan_brace_quantifier(idx)
|
|
301
|
+
return nil unless @src[idx] == "{"
|
|
302
|
+
i = idx + 1
|
|
303
|
+
j = i
|
|
304
|
+
j += 1 while j < @src.length && @src[j].match?(/\d/)
|
|
305
|
+
min_digits = @src[i...j]
|
|
306
|
+
comma = ""
|
|
307
|
+
if @src[j] == ","
|
|
308
|
+
comma = ","
|
|
309
|
+
j += 1
|
|
310
|
+
end
|
|
311
|
+
k = j
|
|
312
|
+
k += 1 while k < @src.length && @src[k].match?(/\d/)
|
|
313
|
+
max_digits = @src[j...k]
|
|
314
|
+
return nil unless @src[k] == "}"
|
|
315
|
+
[min_digits, comma, max_digits, k - idx + 1]
|
|
316
|
+
end
|
|
317
|
+
|
|
318
|
+
def parse_quant
|
|
319
|
+
c = peek
|
|
320
|
+
quant = case c
|
|
321
|
+
when "*" then @pos += 1; { min: 0, max: nil }
|
|
322
|
+
when "+" then @pos += 1; { min: 1, max: nil }
|
|
323
|
+
when "?" then @pos += 1; { min: 0, max: 1 }
|
|
324
|
+
when "{"
|
|
325
|
+
m = scan_brace_quantifier(@pos)
|
|
326
|
+
return nil if m.nil? || (m[0].empty? && m[2].empty?)
|
|
327
|
+
@pos += m[3]
|
|
328
|
+
min = m[0].empty? ? 0 : m[0].to_i
|
|
329
|
+
max = if m[1].empty? then min
|
|
330
|
+
elsif m[2].empty? then nil
|
|
331
|
+
else m[2].to_i
|
|
332
|
+
end
|
|
333
|
+
if min > MAX_REPEAT_COUNT || (max && max > MAX_REPEAT_COUNT)
|
|
334
|
+
refuse("repeat count above #{MAX_REPEAT_COUNT}")
|
|
335
|
+
end
|
|
336
|
+
refuse("repeat range out of order") if max && max < min
|
|
337
|
+
{ min: min, max: max }
|
|
338
|
+
end
|
|
339
|
+
return nil if quant.nil?
|
|
340
|
+
@pos += 1 if peek == "?" || peek == "+"
|
|
341
|
+
refuse("stacked quantifiers") if %w[* + ?].include?(peek) || (peek == "{" && quantifier_at?(@pos))
|
|
342
|
+
quant
|
|
343
|
+
end
|
|
344
|
+
end
|
|
29
345
|
|
|
30
346
|
class << self
|
|
31
347
|
# Validates a regex pattern for potential ReDoS vulnerabilities.
|
|
348
|
+
#
|
|
349
|
+
# Escaped literal text (what `starts_with`, `ends_with`, and
|
|
350
|
+
# `contains` build) always passes; its length cap is
|
|
351
|
+
# `2 * max_length + 4` because escaping can double it. Any other
|
|
352
|
+
# pattern is parsed and refused when it contains a shape that
|
|
353
|
+
# backtracks catastrophically on PCRE:
|
|
354
|
+
#
|
|
355
|
+
# * a group repeated more than once (`+`, `*`, `{n,}`, `{n}` or
|
|
356
|
+
# `{n,m}` with a top above 1) whose body holds a quantifier, an
|
|
357
|
+
# alternation, or a backreference: `(a+)+`, `(a|aa)+`, `((a+))+`,
|
|
358
|
+
# `(a+){2,}`, `(a+){20}`;
|
|
359
|
+
# A repeated group stays allowed when each repeat cannot split its
|
|
360
|
+
# input more than one way: an alternation of fixed literals with
|
|
361
|
+
# distinct first characters (`(foo|bar)+`), or a literal separator
|
|
362
|
+
# next to one quantified atom that cannot match it
|
|
363
|
+
# (`(-[a-z0-9]+)*`, `(\.[\w-]+)+`, `([\w-]+\.)+`);
|
|
364
|
+
# * two adjacent unbounded `.*` / `.+` with more pattern after them;
|
|
365
|
+
# * more than three unbounded or wide quantified atoms in a row with
|
|
366
|
+
# no literal between them that they cannot match
|
|
367
|
+
# (`\d+\d+\d+\d+`, `.*a.*a.*a.*a`); groups without a quantifier are
|
|
368
|
+
# read as part of the surrounding sequence;
|
|
369
|
+
# * a subroutine call (`\g<1>`), whitespace inside a repeat count
|
|
370
|
+
# (`{ 2,}`), which newer PCRE2 reads as a quantifier;
|
|
371
|
+
# * an inline comment `(?#...)`, extended mode (`(?x)` or a Regexp
|
|
372
|
+
# with `Regexp::EXTENDED`), recursion, conditionals, or anything
|
|
373
|
+
# else the reader does not recognize, and unbalanced patterns.
|
|
374
|
+
#
|
|
375
|
+
# Repeats of a single atom (`.{1,255}`, `\d{1,100}`, `x{1000}`, `a+`)
|
|
376
|
+
# and lookarounds without such a shape inside (`^(?!test)`) pass.
|
|
32
377
|
# @param pattern [String, Regexp] the pattern to validate
|
|
33
378
|
# @param max_length [Integer] maximum allowed pattern length
|
|
34
379
|
# @raise [ArgumentError] if the pattern is potentially dangerous
|
|
35
380
|
# @return [String] the validated pattern string
|
|
36
381
|
def validate!(pattern, max_length: MAX_PATTERN_LENGTH)
|
|
382
|
+
if pattern.is_a?(Regexp) && (pattern.options & Regexp::EXTENDED) != 0
|
|
383
|
+
raise ArgumentError, "Regex pattern uses extended mode (the x flag), which is not allowed: " \
|
|
384
|
+
"it can hide quantifiers in comments. Pattern: #{pattern.source.inspect}"
|
|
385
|
+
end
|
|
37
386
|
pattern_str = pattern.is_a?(Regexp) ? pattern.source : pattern.to_s
|
|
387
|
+
literal = literal_pattern?(pattern_str)
|
|
388
|
+
cap = literal ? (2 * max_length) + 4 : max_length
|
|
38
389
|
|
|
39
|
-
if pattern_str.length >
|
|
40
|
-
raise ArgumentError, "Regex pattern too long (#{pattern_str.length} chars, max #{
|
|
390
|
+
if pattern_str.length > cap
|
|
391
|
+
raise ArgumentError, "Regex pattern too long (#{pattern_str.length} chars, max #{cap}). " \
|
|
41
392
|
"Long patterns can cause performance issues."
|
|
42
393
|
end
|
|
43
394
|
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
395
|
+
return pattern_str if literal
|
|
396
|
+
|
|
397
|
+
reason = begin
|
|
398
|
+
check_tree(Parser.new(pattern_str).parse)
|
|
399
|
+
rescue Refused => e
|
|
400
|
+
e.message
|
|
48
401
|
end
|
|
402
|
+
if reason
|
|
403
|
+
raise ArgumentError, "Regex pattern contains potentially dangerous constructs that could cause " \
|
|
404
|
+
"ReDoS (Regular Expression Denial of Service): #{reason}. Pattern: #{pattern_str.inspect}"
|
|
49
405
|
end
|
|
50
406
|
|
|
51
407
|
pattern_str
|
|
52
408
|
end
|
|
53
409
|
|
|
410
|
+
# Whether a pattern is escaped literal text, optionally anchored
|
|
411
|
+
# (`^text`, `text$`) or wrapped in `.*` (`.*text.*`). A pattern that is
|
|
412
|
+
# only `.*` wrappers (`.*.*`) is not literal text.
|
|
413
|
+
# @param pattern_str [String]
|
|
414
|
+
# @return [Boolean]
|
|
415
|
+
def literal_pattern?(pattern_str)
|
|
416
|
+
body = pattern_str.dup
|
|
417
|
+
dot_prefix = false
|
|
418
|
+
if body.start_with?("^")
|
|
419
|
+
body = body[1..]
|
|
420
|
+
elsif body.start_with?(".*")
|
|
421
|
+
body = body[2..]
|
|
422
|
+
dot_prefix = true
|
|
423
|
+
end
|
|
424
|
+
dot_suffix = false
|
|
425
|
+
if body.end_with?(".*") && !body.end_with?("\\.*")
|
|
426
|
+
body = body[0..-3]
|
|
427
|
+
dot_suffix = true
|
|
428
|
+
elsif body.end_with?("$") && !body.end_with?("\\$")
|
|
429
|
+
body = body[0..-2]
|
|
430
|
+
end
|
|
431
|
+
return false if dot_prefix && dot_suffix && body.empty?
|
|
432
|
+
LITERAL_BODY.match?(body)
|
|
433
|
+
end
|
|
434
|
+
|
|
435
|
+
# Validates `$options` flags against {ALLOWED_OPTIONS}.
|
|
436
|
+
# @param options [Object]
|
|
437
|
+
# @raise [ArgumentError] on a non-String or an unknown flag.
|
|
438
|
+
def validate_options!(options)
|
|
439
|
+
unless options.is_a?(String)
|
|
440
|
+
raise ArgumentError, "Regex $options must be a String (got #{options.class})."
|
|
441
|
+
end
|
|
442
|
+
bad = options.chars.uniq.reject { |c| ALLOWED_OPTIONS.include?(c) }
|
|
443
|
+
unless bad.empty?
|
|
444
|
+
raise ArgumentError, "Regex $options contains unsupported flags #{bad.join.inspect}. " \
|
|
445
|
+
"Allowed: #{ALLOWED_OPTIONS.chars.join(", ")}."
|
|
446
|
+
end
|
|
447
|
+
options
|
|
448
|
+
end
|
|
449
|
+
|
|
450
|
+
# Validates every `$regex` (and its `$options`) inside a compiled where
|
|
451
|
+
# clause, at any depth: field values, `$not` / `$elemMatch` wrappers,
|
|
452
|
+
# and `$or` / `$and` / `$nor` branches, plus Regexp and BSON regex
|
|
453
|
+
# values anywhere (equality, `$not`, `$in`, `$nin`, `$all`). SDK
|
|
454
|
+
# routing markers (`__` keys) are skipped. Literal patterns built from
|
|
455
|
+
# escaped input pass.
|
|
456
|
+
# @param node [Object] a compiled where clause or part of one.
|
|
457
|
+
# @raise [ArgumentError] when a pattern or its options are unsafe.
|
|
458
|
+
# @return [void]
|
|
459
|
+
def validate_where!(node)
|
|
460
|
+
case node
|
|
461
|
+
when Hash
|
|
462
|
+
node.each do |key, value|
|
|
463
|
+
key_str = key.to_s
|
|
464
|
+
next if key_str.start_with?("__")
|
|
465
|
+
if key_str == "$regex"
|
|
466
|
+
value.is_a?(String) ? validate!(value) : validate_where!(value)
|
|
467
|
+
elsif key_str == "$options" && (node.key?("$regex") || node.key?(:$regex))
|
|
468
|
+
validate_options!(value)
|
|
469
|
+
else
|
|
470
|
+
validate_where!(value)
|
|
471
|
+
end
|
|
472
|
+
end
|
|
473
|
+
when Array
|
|
474
|
+
node.each { |item| validate_where!(item) }
|
|
475
|
+
when Regexp
|
|
476
|
+
# A Regexp value (equality, `$not`, `$in`) is a regex match too.
|
|
477
|
+
validate!(node)
|
|
478
|
+
else
|
|
479
|
+
if defined?(BSON::Regexp::Raw) && node.is_a?(BSON::Regexp::Raw)
|
|
480
|
+
if node.options.to_s.include?("x")
|
|
481
|
+
raise ArgumentError, "Regex pattern uses extended mode (the x flag), which is not allowed."
|
|
482
|
+
end
|
|
483
|
+
validate!(node.pattern.to_s)
|
|
484
|
+
end
|
|
485
|
+
end
|
|
486
|
+
nil
|
|
487
|
+
end
|
|
488
|
+
|
|
489
|
+
# @!visibility private
|
|
490
|
+
# The first refused shape in a parsed pattern, or nil.
|
|
491
|
+
def check_tree(node)
|
|
492
|
+
case node.first
|
|
493
|
+
when :alt
|
|
494
|
+
node[1].each do |seq|
|
|
495
|
+
reason = check_tree(seq)
|
|
496
|
+
return reason if reason
|
|
497
|
+
end
|
|
498
|
+
when :seq
|
|
499
|
+
items = node[1]
|
|
500
|
+
reason = wide_run_reason(flatten_items(items))
|
|
501
|
+
return reason if reason
|
|
502
|
+
items.each_with_index do |item, idx|
|
|
503
|
+
atom = item[:atom]
|
|
504
|
+
quant = item[:quant]
|
|
505
|
+
if atom.first == :group
|
|
506
|
+
if repeats?(quant) && complex?(atom[1]) && !separated_repeat?(atom[1])
|
|
507
|
+
return "a repeated group contains a quantifier, alternation, or backreference"
|
|
508
|
+
end
|
|
509
|
+
reason = check_tree(atom[1])
|
|
510
|
+
return reason if reason
|
|
511
|
+
end
|
|
512
|
+
nxt = items[idx + 1]
|
|
513
|
+
if unbounded_dot?(item) && nxt && unbounded_dot?(nxt) &&
|
|
514
|
+
!items[(idx + 2)..].all? { |rest| rest[:atom].first == :anchor }
|
|
515
|
+
return "adjacent unbounded .* or .+ followed by more pattern"
|
|
516
|
+
end
|
|
517
|
+
end
|
|
518
|
+
end
|
|
519
|
+
nil
|
|
520
|
+
end
|
|
521
|
+
private :check_tree
|
|
522
|
+
|
|
523
|
+
# @!visibility private
|
|
524
|
+
# The sequence with non-repeated, non-lookaround groups spliced in,
|
|
525
|
+
# so `(?:.*)(?:.*)` reads as `.*.*`.
|
|
526
|
+
def flatten_items(items)
|
|
527
|
+
items.flat_map do |item|
|
|
528
|
+
atom = item[:atom]
|
|
529
|
+
if atom.first == :group && item[:quant].nil? && atom[2] != :lookaround && atom[1][1].length == 1
|
|
530
|
+
flatten_items(atom[1][1].first[1])
|
|
531
|
+
else
|
|
532
|
+
[item]
|
|
533
|
+
end
|
|
534
|
+
end
|
|
535
|
+
end
|
|
536
|
+
private :flatten_items
|
|
537
|
+
|
|
538
|
+
# @!visibility private
|
|
539
|
+
def wide?(item)
|
|
540
|
+
quant = item[:quant]
|
|
541
|
+
!quant.nil? && (quant[:max].nil? || quant[:max] - quant[:min] > WIDE_RANGE)
|
|
542
|
+
end
|
|
543
|
+
private :wide?
|
|
544
|
+
|
|
545
|
+
# @!visibility private
|
|
546
|
+
# A refusal reason when more than {MAX_WIDE_RUN} wide atoms appear
|
|
547
|
+
# without a literal character between them that none of the wide
|
|
548
|
+
# atoms in the sequence can match.
|
|
549
|
+
def wide_run_reason(items)
|
|
550
|
+
wide_items = items.select { |item| wide?(item) }
|
|
551
|
+
return nil if wide_items.length <= MAX_WIDE_RUN
|
|
552
|
+
count = 0
|
|
553
|
+
items.each do |item|
|
|
554
|
+
if wide?(item)
|
|
555
|
+
count += 1
|
|
556
|
+
return "more than #{MAX_WIDE_RUN} unbounded or wide quantifiers in a row" if count > MAX_WIDE_RUN
|
|
557
|
+
elsif (ch = literal_char(item)) && wide_items.none? { |w| atom_matches?(w, ch) }
|
|
558
|
+
count = 0
|
|
559
|
+
end
|
|
560
|
+
end
|
|
561
|
+
nil
|
|
562
|
+
end
|
|
563
|
+
private :wide_run_reason
|
|
564
|
+
|
|
565
|
+
# @!visibility private
|
|
566
|
+
# The character a single unquantified literal atom matches (`a`, `-`,
|
|
567
|
+
# `\.`), or nil.
|
|
568
|
+
def literal_char(item)
|
|
569
|
+
return nil unless item[:quant].nil? && item[:atom] == [:char]
|
|
570
|
+
text = item[:text].to_s
|
|
571
|
+
if text.length == 1 && !text.match?(/[\\.\[\]()|?*+{}^$]/)
|
|
572
|
+
text
|
|
573
|
+
elsif (m = text.match(/\A\\([^A-Za-z0-9])\z/))
|
|
574
|
+
m[1]
|
|
575
|
+
end
|
|
576
|
+
end
|
|
577
|
+
private :literal_char
|
|
578
|
+
|
|
579
|
+
# @!visibility private
|
|
580
|
+
# Whether an atom (ignoring its quantifier) can match `ch` in either
|
|
581
|
+
# case. Anything that cannot be checked counts as a match.
|
|
582
|
+
def atom_matches?(item, ch)
|
|
583
|
+
atom = item[:atom]
|
|
584
|
+
return true unless atom == [:char] || atom == [:dot]
|
|
585
|
+
re = begin
|
|
586
|
+
Regexp.new(atom == [:dot] ? "." : item[:text].to_s)
|
|
587
|
+
rescue RegexpError, ArgumentError
|
|
588
|
+
nil
|
|
589
|
+
end
|
|
590
|
+
return true if re.nil?
|
|
591
|
+
[ch, ch.swapcase, ch.downcase(:fold), ch.upcase].uniq.any? { |c| re.match?(c) }
|
|
592
|
+
end
|
|
593
|
+
private :atom_matches?
|
|
594
|
+
|
|
595
|
+
# @!visibility private
|
|
596
|
+
# A repeated group body that cannot split its input more than one way
|
|
597
|
+
# per repeat, so repeating it is safe:
|
|
598
|
+
#
|
|
599
|
+
# * an alternation of fixed literal strings whose first characters
|
|
600
|
+
# differ, compared case-insensitively because the pattern may run
|
|
601
|
+
# with the `i` option or `(?i)` (`(foo|bar)+`, but not `(a|Aa)+`);
|
|
602
|
+
# * a mandatory literal separator next to one quantified atom that
|
|
603
|
+
# cannot match it (`(-[a-z0-9]+)*`, `(\.[\w-]+)+`, `([\w-]+\.)+`).
|
|
604
|
+
def separated_repeat?(body)
|
|
605
|
+
branches = body[1]
|
|
606
|
+
if branches.length > 1
|
|
607
|
+
firsts = branches.map do |seq|
|
|
608
|
+
items = seq[1]
|
|
609
|
+
return false if items.empty?
|
|
610
|
+
chars = items.map { |item| literal_char(item) }
|
|
611
|
+
return false if chars.any?(&:nil?)
|
|
612
|
+
chars.first.downcase(:fold)
|
|
613
|
+
end
|
|
614
|
+
return firsts.uniq.length == firsts.length
|
|
615
|
+
end
|
|
616
|
+
items = branches.first[1]
|
|
617
|
+
return false unless items.length == 2
|
|
618
|
+
sep, run = items
|
|
619
|
+
sep, run = run, sep if literal_char(sep).nil?
|
|
620
|
+
ch = literal_char(sep)
|
|
621
|
+
return false if ch.nil?
|
|
622
|
+
return false unless run[:quant] && (run[:atom] == [:char] || run[:atom] == [:dot])
|
|
623
|
+
!atom_matches?(run, ch)
|
|
624
|
+
end
|
|
625
|
+
private :separated_repeat?
|
|
626
|
+
|
|
627
|
+
# @!visibility private
|
|
628
|
+
def repeats?(quant)
|
|
629
|
+
!quant.nil? && (quant[:max].nil? || quant[:max] > 1)
|
|
630
|
+
end
|
|
631
|
+
private :repeats?
|
|
632
|
+
|
|
633
|
+
# @!visibility private
|
|
634
|
+
def unbounded_dot?(item)
|
|
635
|
+
item[:atom].first == :dot && item[:quant] && item[:quant][:max].nil?
|
|
636
|
+
end
|
|
637
|
+
private :unbounded_dot?
|
|
638
|
+
|
|
639
|
+
# @!visibility private
|
|
640
|
+
# Whether a quantifier lets the engine choose how many times to match
|
|
641
|
+
# (`?`, `*`, `+`, `{0,n}`, `{n,m}` with m > n, `{n,}`), so each repeat
|
|
642
|
+
# of an enclosing group can split the input another way. A fixed count
|
|
643
|
+
# (`{3}`) gives no choice.
|
|
644
|
+
def branches?(quant)
|
|
645
|
+
!quant.nil? && (quant[:max].nil? || quant[:max] != quant[:min])
|
|
646
|
+
end
|
|
647
|
+
private :branches?
|
|
648
|
+
|
|
649
|
+
# @!visibility private
|
|
650
|
+
# Whether a subtree holds a quantifier that introduces a choice
|
|
651
|
+
# (including an optional `?`, which matches zero or one time), an
|
|
652
|
+
# alternation with more than one branch, or a backreference. Any of
|
|
653
|
+
# these inside a group repeated more than once multiplies the ways a
|
|
654
|
+
# failing match can be retried: `(a?){100}a{100}` backtracks as badly
|
|
655
|
+
# as `(a+)+`.
|
|
656
|
+
def complex?(node)
|
|
657
|
+
case node.first
|
|
658
|
+
when :alt
|
|
659
|
+
return true if node[1].length > 1
|
|
660
|
+
node[1].any? { |seq| complex?(seq) }
|
|
661
|
+
when :seq
|
|
662
|
+
node[1].any? do |item|
|
|
663
|
+
atom = item[:atom]
|
|
664
|
+
branches?(item[:quant]) ||
|
|
665
|
+
atom.first == :backref ||
|
|
666
|
+
(atom.first == :group && complex?(atom[1]))
|
|
667
|
+
end
|
|
668
|
+
else
|
|
669
|
+
false
|
|
670
|
+
end
|
|
671
|
+
end
|
|
672
|
+
private :complex?
|
|
673
|
+
|
|
54
674
|
# Checks if a pattern is safe without raising an exception.
|
|
55
675
|
# @param pattern [String, Regexp] the pattern to check
|
|
56
676
|
# @return [Boolean] true if safe, false if potentially dangerous
|