database_consistency 3.0.12 → 3.0.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/database_consistency/helper.rb +288 -49
- data/lib/database_consistency/version.rb +1 -1
- metadata +2 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: c0599ad1c39057b6a4ec4bb2c60938454a92296906258cda66b5b9f47081447e
|
|
4
|
+
data.tar.gz: 77d57bb76d65535e3dea3dfa88d8b5efcce14a230682ee472fd26a1f605c619a
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 9cffcfa84c2731c584788352910c712f9947cc4fcdc73f24bb68128889f6ec0674accdef8d2390d113fb4f16c7e37e5c859dd09fe649d75b2be04b4964fc10c3
|
|
7
|
+
data.tar.gz: 3fc92238031f7e9e3611179a5fcad66b8105374eaa799416ae426248a9f88b7baafd61642e98a2eb03d3db0d591cf241d7515388f309b058dca2c20fd04d783d
|
|
@@ -194,29 +194,165 @@ module DatabaseConsistency
|
|
|
194
194
|
# for a column name by the boolean-predicate normalizer.
|
|
195
195
|
LITERAL_PLACEHOLDER = '__DATABASE_CONSISTENCY_LITERAL<%<index>d>__'
|
|
196
196
|
|
|
197
|
+
# Matches one single-quoted string literal, including any `''` it contains:
|
|
198
|
+
# SQL escapes a quote by doubling it, so a `''` pair is part of the value
|
|
199
|
+
# rather than the end of it.
|
|
200
|
+
CONDITION_LITERAL = /'(?:[^']|'')*'/.freeze
|
|
201
|
+
|
|
202
|
+
# Matches a masked literal, so steps that run on masked SQL can step over
|
|
203
|
+
# the `<` and `>` in the placeholder.
|
|
204
|
+
MASKED_LITERAL = Regexp.new(
|
|
205
|
+
Regexp.escape(LITERAL_PLACEHOLDER).sub('%<index>d') { '\d+' }
|
|
206
|
+
).freeze
|
|
207
|
+
|
|
208
|
+
# Matches an operator together with whatever spaces were written around it.
|
|
209
|
+
# `+` and `-` are left out: either can be a sign as well as an operator, and
|
|
210
|
+
# telling the two apart takes a parser. `->` and `->>` are the exception,
|
|
211
|
+
# since neither can be read as a sign.
|
|
212
|
+
CONDITION_OPERATOR = %r{\s*(->>?|[<>=!~@#%^&|?*/]+)\s*}.freeze
|
|
213
|
+
|
|
214
|
+
# Matches either of the two, so the spacing step can find operators while
|
|
215
|
+
# passing over the placeholders.
|
|
216
|
+
MASKED_LITERAL_OR_OPERATOR = Regexp.union(MASKED_LITERAL, CONDITION_OPERATOR).freeze
|
|
217
|
+
|
|
218
|
+
# Matches a number PostgreSQL had to quote in order to coerce it, together
|
|
219
|
+
# with the cast that says it is a number rather than a string. `::text` is
|
|
220
|
+
# deliberately absent from the list so a genuine string keeps its quotes.
|
|
221
|
+
COERCED_NUMERIC_LITERAL = /
|
|
222
|
+
' (-? \d+ (?:\.\d+)? (?: e[+-]?\d+ )? ) '
|
|
223
|
+
(?= :: (?: integer | bigint | numeric | double\s+precision ) \b )
|
|
224
|
+
/xi.freeze
|
|
225
|
+
|
|
226
|
+
# Matches a PostgreSQL cast, covering the type names written as several
|
|
227
|
+
# words, the length or precision an explicit cast carries and the `[]` of an
|
|
228
|
+
# array type: `::text`, `::text[]`, `::double precision`,
|
|
229
|
+
# `::character varying(3)`, `::numeric(5,2)`, `::time without time zone`.
|
|
230
|
+
# A date or time type carries its precision in the middle of its name, as
|
|
231
|
+
# `::timestamp(0) without time zone`, so that branch spells out its own.
|
|
232
|
+
CONDITION_CAST = /
|
|
233
|
+
::
|
|
234
|
+
(?:
|
|
235
|
+
character\s+varying |
|
|
236
|
+
double\s+precision |
|
|
237
|
+
bit\s+varying |
|
|
238
|
+
(?:timestamp|time) (?:\(\d+\))? \s+ (?:with|without)\s+time\s+zone |
|
|
239
|
+
\w+
|
|
240
|
+
)
|
|
241
|
+
(?:\(\d+(?:\s*,\s*\d+)?\))?
|
|
242
|
+
(?:\[\])?
|
|
243
|
+
/xi.freeze
|
|
244
|
+
|
|
245
|
+
# Matches a number written in exponent notation, capturing the sign, the
|
|
246
|
+
# digits on each side of the decimal point and the exponent separately so
|
|
247
|
+
# the point can be shifted through the digits as text. The lookbehind keeps
|
|
248
|
+
# the digits of an identifier such as `a1e5` out of it.
|
|
249
|
+
EXPONENT_LITERAL = /
|
|
250
|
+
(?<![\w.])
|
|
251
|
+
(-?) (\d+) (?: \.(\d+) )? e ([+-]?\d+)
|
|
252
|
+
/xi.freeze
|
|
253
|
+
|
|
254
|
+
# The parentheses right after `IN` or `NOT IN` are the list itself rather
|
|
255
|
+
# than something wrapped around a value, so the patterns below leave them
|
|
256
|
+
# alone and `qty IN (1)` stays a list of one. This covers only the
|
|
257
|
+
# parenthesis that opens the list; a value with parentheses of its own
|
|
258
|
+
# further along it, such as the `(1)` in `qty IN ((1), 2)`, still loses
|
|
259
|
+
# them.
|
|
260
|
+
IN_LIST_OPENING = /(?<!\bIN\s)/i.freeze
|
|
261
|
+
|
|
262
|
+
# Matches a bare identifier wrapped in parentheses, e.g. `(internal_name)`.
|
|
263
|
+
# The lookbehind keeps the argument list of a call such as `lower(name)`
|
|
264
|
+
# intact.
|
|
265
|
+
WRAPPED_IDENTIFIER = /(?<![\w.])#{IN_LIST_OPENING}\(([a-z_][\w.]*)\)/i.freeze
|
|
266
|
+
|
|
267
|
+
# Matches a parenthesized numeric literal, e.g. `(0)` or `(0.001)`, which is
|
|
268
|
+
# what a cast such as `(0)::numeric` leaves behind once the cast is gone.
|
|
269
|
+
# The lookbehind keeps the argument list of a call such as `abs(1)` intact.
|
|
270
|
+
WRAPPED_NUMBER = /(?<![\w.])#{IN_LIST_OPENING}\((-?\d+(?:\.\d+)?(?:e-?\d+)?)\)/.freeze
|
|
271
|
+
|
|
272
|
+
# Matches parentheses wrapping exactly one function call, such as the
|
|
273
|
+
# `(abs(1))` a removed `::numeric` cast leaves behind. The inner group
|
|
274
|
+
# recurses so the call's own argument list may nest, and the lookbehind
|
|
275
|
+
# keeps a call's own parentheses out of it.
|
|
276
|
+
WRAPPED_FUNCTION_CALL = /
|
|
277
|
+
(?<![\w.]) #{IN_LIST_OPENING}
|
|
278
|
+
\( (?<call>[a-z_][\w.]* (?<arguments>\( (?:[^()] | \g<arguments>)* \)) ) \)
|
|
279
|
+
/xi.freeze
|
|
280
|
+
|
|
281
|
+
# Matches a bare negated boolean predicate such as `NOT archived`, in the
|
|
282
|
+
# three places one can stand: at the start of an expression, after `AND` or
|
|
283
|
+
# `OR`, or after an opening parenthesis. The whitespace before whatever
|
|
284
|
+
# follows sits inside the lookahead, so the match leaves it in place instead
|
|
285
|
+
# of consuming it and fusing the next `AND` / `OR` to the rewritten
|
|
286
|
+
# predicate. The lookbehind keeps a call's own parenthesis out of the
|
|
287
|
+
# boolean positions, so the argument of `lower(...)` is not read as a
|
|
288
|
+
# predicate of its own.
|
|
289
|
+
NEGATED_BOOLEAN_PREDICATE = /
|
|
290
|
+
(^ | (?: \bAND\b | \bOR\b | (?<![\w.]) \( ))
|
|
291
|
+
\s* NOT \s+ ([a-z_][\w.]*)
|
|
292
|
+
(?= \s* (?: $ | \bAND\b | \bOR\b | \) ))
|
|
293
|
+
/xi.freeze
|
|
294
|
+
|
|
295
|
+
# Matches a bare boolean predicate such as `most_recent` in those same three
|
|
296
|
+
# places, with the same lookahead and lookbehind. It runs after the negated
|
|
297
|
+
# form so that `NOT archived` is already gone and cannot be read as the
|
|
298
|
+
# predicate `archived`.
|
|
299
|
+
BARE_BOOLEAN_PREDICATE = /
|
|
300
|
+
(^ | (?: \bAND\b | \bOR\b | (?<![\w.]) \( ))
|
|
301
|
+
\s* ([a-z_][\w.]*)
|
|
302
|
+
(?= \s* (?: $ | \bAND\b | \bOR\b | \) ))
|
|
303
|
+
/xi.freeze
|
|
304
|
+
|
|
305
|
+
# Matches `column = ANY (ARRAY[...])` or `column != ALL ((ARRAY[...]))`,
|
|
306
|
+
# capturing the column name, the operator and the array payload. The inner
|
|
307
|
+
# parentheses come from Postgres indexdefs that wrap the array expression
|
|
308
|
+
# before casting; they are optional, but both or neither, so a group
|
|
309
|
+
# enclosing the whole predicate keeps its own.
|
|
310
|
+
ARRAY_MEMBERSHIP_PREDICATE = /
|
|
311
|
+
(?<column>[a-z_][\w.]*)\s*
|
|
312
|
+
(?<operator>=\s*ANY|(?:!=|<>)\s*ALL)\s*
|
|
313
|
+
\( (?: \(ARRAY\[(?<items>.*?)\]\) | ARRAY\[(?<items>.*?)\] ) \)
|
|
314
|
+
/xi.freeze
|
|
315
|
+
|
|
316
|
+
# Matches SQL like `NOT (column = '' OR column IS NULL)`, holding both sides
|
|
317
|
+
# to the same column with the backreference.
|
|
318
|
+
NEGATED_BLANK_OR_NIL_PREDICATE = /
|
|
319
|
+
NOT \s+ \( \s* \(?
|
|
320
|
+
([a-z_][\w.]*) \s* = \s* '' \s+ OR \s+ \1 \s+ IS \s+ NULL
|
|
321
|
+
\)? \s* \)
|
|
322
|
+
/xi.freeze
|
|
323
|
+
|
|
197
324
|
# Normalizes SQL predicates into a canonical form so semantically equivalent
|
|
198
325
|
# Rails validators and database partial indexes can be compared safely.
|
|
199
326
|
def normalize_condition_sql(sql)
|
|
327
|
+
# The two steps that read the inside of a literal run first, while it is
|
|
328
|
+
# still there to read. Everything after masking works on the shape of the
|
|
329
|
+
# predicate alone and so cannot rewrite a value by accident.
|
|
200
330
|
masked_sql, literals = sql.to_s
|
|
201
|
-
.then { |value|
|
|
202
|
-
.then { |value|
|
|
331
|
+
.then { |value| unquote_numeric_literals(value) }
|
|
332
|
+
.then { |value| normalize_quoted_boolean_literals(value) }
|
|
203
333
|
.then { |value| mask_condition_literals(value) }
|
|
204
334
|
|
|
205
|
-
normalize_masked_condition_sql(
|
|
335
|
+
normalize_masked_condition_sql(
|
|
336
|
+
masked_sql.then { |value| strip_outer_parentheses(value) }
|
|
337
|
+
.then { |value| normalize_boolean_and_null_keywords(value) },
|
|
338
|
+
literals
|
|
339
|
+
)
|
|
206
340
|
end
|
|
207
341
|
|
|
208
342
|
# Finishes normalization after string literals have been masked: runs the
|
|
209
|
-
# regex-based transforms that must not see inside literals,
|
|
210
|
-
#
|
|
343
|
+
# regex-based transforms that must not see inside literals, applies the
|
|
344
|
+
# final structural clean-ups, and only then restores the literal values.
|
|
345
|
+
# Restoring last protects literal contents from whitespace collapse and
|
|
346
|
+
# clause sorting.
|
|
211
347
|
def normalize_masked_condition_sql(masked_sql, literals)
|
|
212
348
|
masked_sql
|
|
213
|
-
.then { |value|
|
|
349
|
+
.then { |value| normalize_adapter_syntax(value) }
|
|
214
350
|
.then { |value| normalize_boolean_predicates(value) }
|
|
215
351
|
.then { |value| normalize_array_any_predicates(value) }
|
|
216
|
-
.then { |value| unmask_condition_literals(value, literals) }
|
|
217
352
|
.then { |value| normalize_negated_blank_or_nil_predicates(value) }
|
|
218
|
-
.then { |value| sort_and_clauses(value) }
|
|
353
|
+
.then { |value| sort_and_clauses(value, literals) }
|
|
219
354
|
.then { |value| value.gsub(/\s+/, ' ').strip }
|
|
355
|
+
.then { |value| unmask_condition_literals(value, literals) }
|
|
220
356
|
end
|
|
221
357
|
|
|
222
358
|
# Masks non-empty string literals so later regexes cannot rewrite their
|
|
@@ -224,7 +360,7 @@ module DatabaseConsistency
|
|
|
224
360
|
# normalization relies on them.
|
|
225
361
|
def mask_condition_literals(sql)
|
|
226
362
|
literals = []
|
|
227
|
-
masked_sql = sql.gsub(
|
|
363
|
+
masked_sql = sql.gsub(CONDITION_LITERAL) do |match|
|
|
228
364
|
if match == "''"
|
|
229
365
|
match
|
|
230
366
|
else
|
|
@@ -244,9 +380,52 @@ module DatabaseConsistency
|
|
|
244
380
|
sql
|
|
245
381
|
end
|
|
246
382
|
|
|
247
|
-
#
|
|
248
|
-
|
|
383
|
+
# PostgreSQL writes any literal it had to coerce as a quoted string with a
|
|
384
|
+
# cast: `-1` becomes `'-1'::integer`, `-1.5` becomes `'-1.5'::numeric` and
|
|
385
|
+
# `1e+20` becomes `'1e+20'::double precision`. Unquoting those lets them line
|
|
386
|
+
# up with the bare numbers Active Record generates. A `::text` cast is left
|
|
387
|
+
# alone so a genuine string comparison keeps its quotes.
|
|
388
|
+
def unquote_numeric_literals(sql)
|
|
389
|
+
sql.gsub(COERCED_NUMERIC_LITERAL) { Regexp.last_match(1) }
|
|
390
|
+
end
|
|
391
|
+
|
|
392
|
+
# Rewrites a boolean written as the quoted `'t'` / `'f'` PostgreSQL stores.
|
|
393
|
+
# It reads the value inside the quotes, so it has to run before literals are
|
|
394
|
+
# masked, while that value is still there to read.
|
|
395
|
+
def normalize_quoted_boolean_literals(sql)
|
|
396
|
+
# Normalize PostgreSQL boolean literals stored as `'t'` / `'f'` inside
|
|
397
|
+
# comparisons. The operator is allowed to touch or be surrounded by
|
|
398
|
+
# arbitrary whitespace so forms like `flag='t'` and `flag <> 'f'` all
|
|
399
|
+
# collapse to the same canonical shape. Inequality is preserved as `!=`
|
|
400
|
+
# because `flag <> 't'` is not the same as `flag = 'f'` (NULL handling
|
|
401
|
+
# differs), so they must not share a canonical form. The lookbehind holds
|
|
402
|
+
# the equality patterns to a standalone `=`, so the ordering comparison in
|
|
403
|
+
# `note >= 't'` keeps both its operator and its value.
|
|
404
|
+
sql
|
|
405
|
+
.gsub(/(?<![<>!])\s*=\s*'t'/, ' = 1')
|
|
406
|
+
.gsub(/(?<![<>!])\s*=\s*'f'/, ' = 0')
|
|
407
|
+
.gsub(/\s*<>\s*'t'/, ' != 1')
|
|
408
|
+
.gsub(/\s*<>\s*'f'/, ' != 0')
|
|
409
|
+
.gsub(/\s*!=\s*'t'/, ' != 1')
|
|
410
|
+
.gsub(/\s*!=\s*'f'/, ' != 0')
|
|
411
|
+
end
|
|
412
|
+
|
|
413
|
+
# Rewrites the `TRUE` / `FALSE` / `NULL` keywords and the `IS` phrasings
|
|
414
|
+
# around them to one spelling. These run once literals are masked, so a
|
|
415
|
+
# value that happens to read `IS TRUE` keeps its own text.
|
|
416
|
+
def normalize_boolean_and_null_keywords(sql)
|
|
249
417
|
normalized_sql = sql.dup
|
|
418
|
+
# `IS NOT TRUE` / `IS NOT FALSE` are matched before the bare `IS TRUE` /
|
|
419
|
+
# `IS FALSE` forms so the longer phrase wins. They normalize to `IS NOT 1`
|
|
420
|
+
# / `IS NOT 0` rather than `= 0` / `= 1` because `IS NOT TRUE` is not the
|
|
421
|
+
# same as `= FALSE` (NULL handling differs).
|
|
422
|
+
normalized_sql = normalized_sql.gsub(/\bIS\s+NOT\s+TRUE\b/i, ' IS NOT 1')
|
|
423
|
+
normalized_sql = normalized_sql.gsub(/\bIS\s+NOT\s+FALSE\b/i, ' IS NOT 0')
|
|
424
|
+
# `/\bIS\s+TRUE\b/i` and `/\bIS\s+FALSE\b/i` normalize predicate forms
|
|
425
|
+
# like `flag IS TRUE` to `flag = 1` so they match `flag = TRUE` and
|
|
426
|
+
# `flag = 't'`.
|
|
427
|
+
normalized_sql = normalized_sql.gsub(/\bIS\s+TRUE\b/i, ' = 1')
|
|
428
|
+
normalized_sql = normalized_sql.gsub(/\bIS\s+FALSE\b/i, ' = 0')
|
|
250
429
|
# `/\bTRUE\b/i` and `/\bFALSE\b/i` normalize boolean literals to `1` / `0`
|
|
251
430
|
# so they match SQL generated by Active Record on some adapters.
|
|
252
431
|
normalized_sql = normalized_sql.gsub(/\bTRUE\b/i, '1').gsub(/\bFALSE\b/i, '0')
|
|
@@ -254,27 +433,88 @@ module DatabaseConsistency
|
|
|
254
433
|
normalized_sql = normalized_sql.gsub(/\bIS\s+NOT\s+NULL\b/i, ' IS NOT NULL')
|
|
255
434
|
# `/\bIS\s+NULL\b/i` normalizes `IS NULL` spacing and casing.
|
|
256
435
|
normalized_sql = normalized_sql.gsub(/\bIS\s+NULL\b/i, ' IS NULL')
|
|
257
|
-
# `/ = 't'/` and `/ = 'f'/` normalize PostgreSQL boolean literals stored
|
|
258
|
-
# as `'t'` / `'f'` inside comparisons.
|
|
259
|
-
normalized_sql = normalized_sql.gsub(/ = 't'/, ' = 1').gsub(/ = 'f'/, ' = 0')
|
|
260
436
|
normalized_sql.gsub(/\s+/, ' ').strip
|
|
261
437
|
end
|
|
262
438
|
|
|
263
|
-
#
|
|
264
|
-
|
|
439
|
+
# Rewrites exponent notation as the plain decimal PostgreSQL itself writes
|
|
440
|
+
# when it expands a literal, so `1e+20` and the `1.0e+20` Active Record
|
|
441
|
+
# generates reach the same string. The digits are shifted as text rather
|
|
442
|
+
# than through a float, so a wide value keeps every one of them.
|
|
443
|
+
def expand_exponent_literals(sql)
|
|
444
|
+
sql.gsub(EXPONENT_LITERAL) do
|
|
445
|
+
match = Regexp.last_match
|
|
446
|
+
shift_decimal_point(match[1], "#{match[2]}#{match[3]}", match[2].length + match[4].to_i)
|
|
447
|
+
end
|
|
448
|
+
end
|
|
449
|
+
|
|
450
|
+
# Places the decimal point `position` digits into `digits`, padding with
|
|
451
|
+
# zeros on whichever side falls short and dropping a fraction that ends in
|
|
452
|
+
# them, so `1e-20` and `1.0e-20` land on the same digits. A zero that only
|
|
453
|
+
# holds the decimal point's place goes too, so `0.1e+2` reaches `10`.
|
|
454
|
+
def shift_decimal_point(sign, digits, position)
|
|
455
|
+
expanded =
|
|
456
|
+
if position >= digits.length
|
|
457
|
+
digits + ('0' * (position - digits.length))
|
|
458
|
+
elsif position.positive?
|
|
459
|
+
"#{digits[0...position]}.#{digits[position..]}"
|
|
460
|
+
else
|
|
461
|
+
"0.#{'0' * -position}#{digits}"
|
|
462
|
+
end
|
|
463
|
+
# On Ruby < 3.0, frozen strings forbid `sub!`.
|
|
464
|
+
expanded = expanded.sub(/\A0+(?=\d)/, '')
|
|
465
|
+
|
|
466
|
+
"#{sign}#{expanded}".sub(/(\.\d*?)0+\z/, '\1').chomp('.')
|
|
467
|
+
end
|
|
468
|
+
|
|
469
|
+
# Gives every operator a space either side, every comma a space after it
|
|
470
|
+
# and no parenthesis a space on its inside, which is how PostgreSQL writes
|
|
471
|
+
# an indexdef however the index was typed.
|
|
472
|
+
def normalize_operator_spacing(sql)
|
|
473
|
+
spaced_sql = sql.gsub(MASKED_LITERAL_OR_OPERATOR) do |match|
|
|
474
|
+
match.match?(MASKED_LITERAL) ? match : " #{match.strip} "
|
|
475
|
+
end
|
|
476
|
+
spaced_sql = spaced_sql.gsub(/\s*,\s*/, ', ')
|
|
477
|
+
spaced_sql.gsub(/\(\s+/, '(').gsub(/\s+\)/, ')')
|
|
478
|
+
end
|
|
479
|
+
|
|
480
|
+
# Rewrites the spellings that differ between adapters, or between what an
|
|
481
|
+
# adapter stores and what Active Record writes: quoted identifiers, casts,
|
|
482
|
+
# exponent notation, the spacing of an `IN` list, of operators and of
|
|
483
|
+
# commas, the parentheses PostgreSQL adds around a cast operand and the `<>`
|
|
484
|
+
# it writes for inequality. Literals are masked throughout, so none of it
|
|
485
|
+
# reaches the inside of a value.
|
|
486
|
+
def normalize_adapter_syntax(sql)
|
|
265
487
|
# Strips quoted identifiers (double quotes on PostgreSQL/SQLite,
|
|
266
488
|
# backticks on MySQL) so the same column normalizes across adapters.
|
|
267
489
|
normalized_sql = sql.gsub(/["`]/, '')
|
|
268
|
-
|
|
269
|
-
normalized_sql = normalized_sql
|
|
270
|
-
#
|
|
271
|
-
#
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
normalized_sql = normalized_sql
|
|
490
|
+
normalized_sql = normalized_sql.gsub(CONDITION_CAST, '')
|
|
491
|
+
normalized_sql = expand_exponent_literals(normalized_sql)
|
|
492
|
+
# Gives `IN` one space before its list, so `qty IN(1)` and `qty IN (1)`
|
|
493
|
+
# reach the same string and the list is recognisable to the unwrappers
|
|
494
|
+
# below. `\b` keeps a call such as `min(1)` out of it.
|
|
495
|
+
normalized_sql = normalized_sql.gsub(/\bIN\s*\(/i, 'IN (')
|
|
496
|
+
normalized_sql = normalize_operator_spacing(normalized_sql)
|
|
497
|
+
normalized_sql = unwrap_redundant_parentheses(normalized_sql)
|
|
498
|
+
# Rewrites the SQL inequality operator `<>` to `!=`; the spacing step
|
|
499
|
+
# has already given it a space either side.
|
|
500
|
+
normalized_sql = normalized_sql.gsub('<>', '!=')
|
|
275
501
|
normalized_sql.gsub(/\s+/, ' ').strip
|
|
276
502
|
end
|
|
277
503
|
|
|
504
|
+
# Removes the parentheses PostgreSQL puts around an operand it had to cast,
|
|
505
|
+
# which are redundant once the cast itself is gone: `(0)::numeric` -> `0`,
|
|
506
|
+
# `((name)::character varying(3))::text` -> `name`, `(abs(1))::numeric` ->
|
|
507
|
+
# `abs(1)`. Each pass repeats because removing one layer can expose another.
|
|
508
|
+
def unwrap_redundant_parentheses(sql)
|
|
509
|
+
normalized_sql = sql.dup
|
|
510
|
+
|
|
511
|
+
true while normalized_sql.gsub!(WRAPPED_IDENTIFIER, '\1')
|
|
512
|
+
true while normalized_sql.gsub!(WRAPPED_NUMBER, '\1')
|
|
513
|
+
true while normalized_sql.gsub!(WRAPPED_FUNCTION_CALL, '\k<call>')
|
|
514
|
+
|
|
515
|
+
normalized_sql
|
|
516
|
+
end
|
|
517
|
+
|
|
278
518
|
# Repeatedly removes one wrapping layer of parentheses when the whole SQL
|
|
279
519
|
# fragment is enclosed, e.g. `((foo))` -> `foo`.
|
|
280
520
|
def strip_outer_parentheses(sql)
|
|
@@ -317,52 +557,51 @@ module DatabaseConsistency
|
|
|
317
557
|
def normalize_boolean_predicates(sql)
|
|
318
558
|
normalized_sql = sql.dup
|
|
319
559
|
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
normalized_sql.gsub!(
|
|
324
|
-
/(^|(?:\bAND\b|\bOR\b|\())\s*NOT\s+([a-z_][\w.]*)\s*(?=$|(?:\bAND\b|\bOR\b|\)))/i
|
|
325
|
-
) { "#{Regexp.last_match(1)} #{Regexp.last_match(2)} = 0" }
|
|
560
|
+
normalized_sql.gsub!(NEGATED_BOOLEAN_PREDICATE) do
|
|
561
|
+
"#{Regexp.last_match(1)} #{Regexp.last_match(2)} = 0"
|
|
562
|
+
end
|
|
326
563
|
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
/(^|(?:\bAND\b|\bOR\b|\())\s*([a-z_][\w.]*)\s*(?=$|(?:\bAND\b|\bOR\b|\)))/i
|
|
331
|
-
) { "#{Regexp.last_match(1)} #{Regexp.last_match(2)} = 1" }
|
|
564
|
+
normalized_sql.gsub!(BARE_BOOLEAN_PREDICATE) do
|
|
565
|
+
"#{Regexp.last_match(1)} #{Regexp.last_match(2)} = 1"
|
|
566
|
+
end
|
|
332
567
|
|
|
333
568
|
normalized_sql.gsub(/\s+/, ' ').strip
|
|
334
569
|
end
|
|
335
570
|
|
|
336
|
-
# Rewrites PostgreSQL's `= ANY (ARRAY[...])`
|
|
337
|
-
#
|
|
571
|
+
# Rewrites PostgreSQL's `= ANY (ARRAY[...])` and `<> ALL (ARRAY[...])` forms
|
|
572
|
+
# into the `IN (...)` and `NOT IN (...)` Active Record generates for arrays.
|
|
573
|
+
# `<>` has already become `!=` by this point in the pipeline.
|
|
338
574
|
def normalize_array_any_predicates(sql)
|
|
339
|
-
sql.gsub(
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
575
|
+
sql.gsub(ARRAY_MEMBERSHIP_PREDICATE) do
|
|
576
|
+
match = Regexp.last_match
|
|
577
|
+
membership = match[:operator].match?(/ANY/i) ? 'IN' : 'NOT IN'
|
|
578
|
+
|
|
579
|
+
"#{match[:column]} #{membership} (#{match[:items].gsub(/\s+/, ' ').strip})"
|
|
580
|
+
end
|
|
344
581
|
end
|
|
345
582
|
|
|
346
583
|
# Rewrites negated "blank or nil" predicates into the same shape used by
|
|
347
584
|
# `allow_blank`-derived guards: `IS NOT NULL AND != ''`.
|
|
348
585
|
def normalize_negated_blank_or_nil_predicates(sql)
|
|
349
|
-
sql.gsub(
|
|
350
|
-
#
|
|
351
|
-
|
|
352
|
-
/NOT\s+\(\s*\(?([a-z_][\w.]*)\s*=\s*''\s+OR\s+\1\s+IS\s+NULL\)?\s*\)/i
|
|
353
|
-
) { "#{Regexp.last_match(1)} IS NOT NULL AND #{Regexp.last_match(1)} != ''" }
|
|
586
|
+
sql.gsub(NEGATED_BLANK_OR_NIL_PREDICATE) do
|
|
587
|
+
"#{Regexp.last_match(1)} IS NOT NULL AND #{Regexp.last_match(1)} != ''"
|
|
588
|
+
end
|
|
354
589
|
end
|
|
355
590
|
|
|
356
591
|
# Sorts simple `AND` clauses so `a AND b` and `b AND a` normalize to the
|
|
357
|
-
# same string before comparison.
|
|
358
|
-
|
|
592
|
+
# same string before comparison. Two clauses can be identical apart from the
|
|
593
|
+
# string each one compares against, and then those strings decide the order,
|
|
594
|
+
# which is why the literals go back in before the sort. A placeholder is
|
|
595
|
+
# numbered by where its literal appeared, so sorting on the placeholders
|
|
596
|
+
# would leave such a pair in whichever order it arrived in.
|
|
597
|
+
def sort_and_clauses(sql, literals)
|
|
359
598
|
# Matches `AND` with surrounding whitespace and splits the expression into
|
|
360
599
|
# comparable clause fragments.
|
|
361
600
|
clauses = sql.split(/\s+AND\s+/i)
|
|
362
601
|
return sql if clauses.length == 1
|
|
363
602
|
|
|
364
603
|
clauses.map! { |clause| strip_outer_parentheses(clause) }
|
|
365
|
-
clauses.
|
|
604
|
+
clauses.sort_by { |clause| unmask_condition_literals(clause, literals) }.join(' AND ')
|
|
366
605
|
end
|
|
367
606
|
|
|
368
607
|
# Builds the implicit SQL guard introduced by validator options that skip
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: database_consistency
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 3.0.
|
|
4
|
+
version: 3.0.14
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Evgeniy Demin
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-09-
|
|
11
|
+
date: 2026-09-30 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: activerecord
|