database_consistency 3.0.12 → 3.0.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: cf61dd8acf18a8e18b559df07d481e8503576cd228ec74190aa54b2a5b11e200
4
- data.tar.gz: 4c2ffed1815099f0c2102043b976a85236017b6c4d92287741a48459824a047c
3
+ metadata.gz: c0599ad1c39057b6a4ec4bb2c60938454a92296906258cda66b5b9f47081447e
4
+ data.tar.gz: 77d57bb76d65535e3dea3dfa88d8b5efcce14a230682ee472fd26a1f605c619a
5
5
  SHA512:
6
- metadata.gz: b04e984adeec1807d37f83b6826d0e4499e95c1bc8b3042b9a078e348de499d1df1d1b2437972eb2899a6716da8f4e498d946629570e761ec3ef59969254ffac
7
- data.tar.gz: c100ad7001fe64ce78b1fed323b0e1c389b4bacfb899a26a215731ca4327f6b026828063ea48ac3a494958c8b59916a94b88577c16d342c6205b69d843b11769
6
+ metadata.gz: 9cffcfa84c2731c584788352910c712f9947cc4fcdc73f24bb68128889f6ec0674accdef8d2390d113fb4f16c7e37e5c859dd09fe649d75b2be04b4964fc10c3
7
+ data.tar.gz: 3fc92238031f7e9e3611179a5fcad66b8105374eaa799416ae426248a9f88b7baafd61642e98a2eb03d3db0d591cf241d7515388f309b058dca2c20fd04d783d
@@ -194,29 +194,165 @@ module DatabaseConsistency
194
194
  # for a column name by the boolean-predicate normalizer.
195
195
  LITERAL_PLACEHOLDER = '__DATABASE_CONSISTENCY_LITERAL<%<index>d>__'
196
196
 
197
+ # Matches one single-quoted string literal, including any `''` it contains:
198
+ # SQL escapes a quote by doubling it, so a `''` pair is part of the value
199
+ # rather than the end of it.
200
+ CONDITION_LITERAL = /'(?:[^']|'')*'/.freeze
201
+
202
+ # Matches a masked literal, so steps that run on masked SQL can step over
203
+ # the `<` and `>` in the placeholder.
204
+ MASKED_LITERAL = Regexp.new(
205
+ Regexp.escape(LITERAL_PLACEHOLDER).sub('%<index>d') { '\d+' }
206
+ ).freeze
207
+
208
+ # Matches an operator together with whatever spaces were written around it.
209
+ # `+` and `-` are left out: either can be a sign as well as an operator, and
210
+ # telling the two apart takes a parser. `->` and `->>` are the exception,
211
+ # since neither can be read as a sign.
212
+ CONDITION_OPERATOR = %r{\s*(->>?|[<>=!~@#%^&|?*/]+)\s*}.freeze
213
+
214
+ # Matches either of the two, so the spacing step can find operators while
215
+ # passing over the placeholders.
216
+ MASKED_LITERAL_OR_OPERATOR = Regexp.union(MASKED_LITERAL, CONDITION_OPERATOR).freeze
217
+
218
+ # Matches a number PostgreSQL had to quote in order to coerce it, together
219
+ # with the cast that says it is a number rather than a string. `::text` is
220
+ # deliberately absent from the list so a genuine string keeps its quotes.
221
+ COERCED_NUMERIC_LITERAL = /
222
+ ' (-? \d+ (?:\.\d+)? (?: e[+-]?\d+ )? ) '
223
+ (?= :: (?: integer | bigint | numeric | double\s+precision ) \b )
224
+ /xi.freeze
225
+
226
+ # Matches a PostgreSQL cast, covering the type names written as several
227
+ # words, the length or precision an explicit cast carries and the `[]` of an
228
+ # array type: `::text`, `::text[]`, `::double precision`,
229
+ # `::character varying(3)`, `::numeric(5,2)`, `::time without time zone`.
230
+ # A date or time type carries its precision in the middle of its name, as
231
+ # `::timestamp(0) without time zone`, so that branch spells out its own.
232
+ CONDITION_CAST = /
233
+ ::
234
+ (?:
235
+ character\s+varying |
236
+ double\s+precision |
237
+ bit\s+varying |
238
+ (?:timestamp|time) (?:\(\d+\))? \s+ (?:with|without)\s+time\s+zone |
239
+ \w+
240
+ )
241
+ (?:\(\d+(?:\s*,\s*\d+)?\))?
242
+ (?:\[\])?
243
+ /xi.freeze
244
+
245
+ # Matches a number written in exponent notation, capturing the sign, the
246
+ # digits on each side of the decimal point and the exponent separately so
247
+ # the point can be shifted through the digits as text. The lookbehind keeps
248
+ # the digits of an identifier such as `a1e5` out of it.
249
+ EXPONENT_LITERAL = /
250
+ (?<![\w.])
251
+ (-?) (\d+) (?: \.(\d+) )? e ([+-]?\d+)
252
+ /xi.freeze
253
+
254
+ # The parentheses right after `IN` or `NOT IN` are the list itself rather
255
+ # than something wrapped around a value, so the patterns below leave them
256
+ # alone and `qty IN (1)` stays a list of one. This covers only the
257
+ # parenthesis that opens the list; a value with parentheses of its own
258
+ # further along it, such as the `(1)` in `qty IN ((1), 2)`, still loses
259
+ # them.
260
+ IN_LIST_OPENING = /(?<!\bIN\s)/i.freeze
261
+
262
+ # Matches a bare identifier wrapped in parentheses, e.g. `(internal_name)`.
263
+ # The lookbehind keeps the argument list of a call such as `lower(name)`
264
+ # intact.
265
+ WRAPPED_IDENTIFIER = /(?<![\w.])#{IN_LIST_OPENING}\(([a-z_][\w.]*)\)/i.freeze
266
+
267
+ # Matches a parenthesized numeric literal, e.g. `(0)` or `(0.001)`, which is
268
+ # what a cast such as `(0)::numeric` leaves behind once the cast is gone.
269
+ # The lookbehind keeps the argument list of a call such as `abs(1)` intact.
270
+ WRAPPED_NUMBER = /(?<![\w.])#{IN_LIST_OPENING}\((-?\d+(?:\.\d+)?(?:e-?\d+)?)\)/.freeze
271
+
272
+ # Matches parentheses wrapping exactly one function call, such as the
273
+ # `(abs(1))` a removed `::numeric` cast leaves behind. The inner group
274
+ # recurses so the call's own argument list may nest, and the lookbehind
275
+ # keeps a call's own parentheses out of it.
276
+ WRAPPED_FUNCTION_CALL = /
277
+ (?<![\w.]) #{IN_LIST_OPENING}
278
+ \( (?<call>[a-z_][\w.]* (?<arguments>\( (?:[^()] | \g<arguments>)* \)) ) \)
279
+ /xi.freeze
280
+
281
+ # Matches a bare negated boolean predicate such as `NOT archived`, in the
282
+ # three places one can stand: at the start of an expression, after `AND` or
283
+ # `OR`, or after an opening parenthesis. The whitespace before whatever
284
+ # follows sits inside the lookahead, so the match leaves it in place instead
285
+ # of consuming it and fusing the next `AND` / `OR` to the rewritten
286
+ # predicate. The lookbehind keeps a call's own parenthesis out of the
287
+ # boolean positions, so the argument of `lower(...)` is not read as a
288
+ # predicate of its own.
289
+ NEGATED_BOOLEAN_PREDICATE = /
290
+ (^ | (?: \bAND\b | \bOR\b | (?<![\w.]) \( ))
291
+ \s* NOT \s+ ([a-z_][\w.]*)
292
+ (?= \s* (?: $ | \bAND\b | \bOR\b | \) ))
293
+ /xi.freeze
294
+
295
+ # Matches a bare boolean predicate such as `most_recent` in those same three
296
+ # places, with the same lookahead and lookbehind. It runs after the negated
297
+ # form so that `NOT archived` is already gone and cannot be read as the
298
+ # predicate `archived`.
299
+ BARE_BOOLEAN_PREDICATE = /
300
+ (^ | (?: \bAND\b | \bOR\b | (?<![\w.]) \( ))
301
+ \s* ([a-z_][\w.]*)
302
+ (?= \s* (?: $ | \bAND\b | \bOR\b | \) ))
303
+ /xi.freeze
304
+
305
+ # Matches `column = ANY (ARRAY[...])` or `column != ALL ((ARRAY[...]))`,
306
+ # capturing the column name, the operator and the array payload. The inner
307
+ # parentheses come from Postgres indexdefs that wrap the array expression
308
+ # before casting; they are optional, but both or neither, so a group
309
+ # enclosing the whole predicate keeps its own.
310
+ ARRAY_MEMBERSHIP_PREDICATE = /
311
+ (?<column>[a-z_][\w.]*)\s*
312
+ (?<operator>=\s*ANY|(?:!=|<>)\s*ALL)\s*
313
+ \( (?: \(ARRAY\[(?<items>.*?)\]\) | ARRAY\[(?<items>.*?)\] ) \)
314
+ /xi.freeze
315
+
316
+ # Matches SQL like `NOT (column = '' OR column IS NULL)`, holding both sides
317
+ # to the same column with the backreference.
318
+ NEGATED_BLANK_OR_NIL_PREDICATE = /
319
+ NOT \s+ \( \s* \(?
320
+ ([a-z_][\w.]*) \s* = \s* '' \s+ OR \s+ \1 \s+ IS \s+ NULL
321
+ \)? \s* \)
322
+ /xi.freeze
323
+
197
324
  # Normalizes SQL predicates into a canonical form so semantically equivalent
198
325
  # Rails validators and database partial indexes can be compared safely.
199
326
  def normalize_condition_sql(sql)
327
+ # The two steps that read the inside of a literal run first, while it is
328
+ # still there to read. Everything after masking works on the shape of the
329
+ # predicate alone and so cannot rewrite a value by accident.
200
330
  masked_sql, literals = sql.to_s
201
- .then { |value| strip_outer_parentheses(value) }
202
- .then { |value| normalize_sql_pre_mask(value) }
331
+ .then { |value| unquote_numeric_literals(value) }
332
+ .then { |value| normalize_quoted_boolean_literals(value) }
203
333
  .then { |value| mask_condition_literals(value) }
204
334
 
205
- normalize_masked_condition_sql(masked_sql, literals)
335
+ normalize_masked_condition_sql(
336
+ masked_sql.then { |value| strip_outer_parentheses(value) }
337
+ .then { |value| normalize_boolean_and_null_keywords(value) },
338
+ literals
339
+ )
206
340
  end
207
341
 
208
342
  # Finishes normalization after string literals have been masked: runs the
209
- # regex-based transforms that must not see inside literals, restores the
210
- # literals, then applies the final clean-ups.
343
+ # regex-based transforms that must not see inside literals, applies the
344
+ # final structural clean-ups, and only then restores the literal values.
345
+ # Restoring last protects literal contents from whitespace collapse and
346
+ # clause sorting.
211
347
  def normalize_masked_condition_sql(masked_sql, literals)
212
348
  masked_sql
213
- .then { |value| normalize_sql_post_mask(value) }
349
+ .then { |value| normalize_adapter_syntax(value) }
214
350
  .then { |value| normalize_boolean_predicates(value) }
215
351
  .then { |value| normalize_array_any_predicates(value) }
216
- .then { |value| unmask_condition_literals(value, literals) }
217
352
  .then { |value| normalize_negated_blank_or_nil_predicates(value) }
218
- .then { |value| sort_and_clauses(value) }
353
+ .then { |value| sort_and_clauses(value, literals) }
219
354
  .then { |value| value.gsub(/\s+/, ' ').strip }
355
+ .then { |value| unmask_condition_literals(value, literals) }
220
356
  end
221
357
 
222
358
  # Masks non-empty string literals so later regexes cannot rewrite their
@@ -224,7 +360,7 @@ module DatabaseConsistency
224
360
  # normalization relies on them.
225
361
  def mask_condition_literals(sql)
226
362
  literals = []
227
- masked_sql = sql.gsub(/'(?:[^']|'')*'/) do |match|
363
+ masked_sql = sql.gsub(CONDITION_LITERAL) do |match|
228
364
  if match == "''"
229
365
  match
230
366
  else
@@ -244,9 +380,52 @@ module DatabaseConsistency
244
380
  sql
245
381
  end
246
382
 
247
- # Normalizations that must run before string literals are masked.
248
- def normalize_sql_pre_mask(sql)
383
+ # PostgreSQL writes any literal it had to coerce as a quoted string with a
384
+ # cast: `-1` becomes `'-1'::integer`, `-1.5` becomes `'-1.5'::numeric` and
385
+ # `1e+20` becomes `'1e+20'::double precision`. Unquoting those lets them line
386
+ # up with the bare numbers Active Record generates. A `::text` cast is left
387
+ # alone so a genuine string comparison keeps its quotes.
388
+ def unquote_numeric_literals(sql)
389
+ sql.gsub(COERCED_NUMERIC_LITERAL) { Regexp.last_match(1) }
390
+ end
391
+
392
+ # Rewrites a boolean written as the quoted `'t'` / `'f'` PostgreSQL stores.
393
+ # It reads the value inside the quotes, so it has to run before literals are
394
+ # masked, while that value is still there to read.
395
+ def normalize_quoted_boolean_literals(sql)
396
+ # Normalize PostgreSQL boolean literals stored as `'t'` / `'f'` inside
397
+ # comparisons. The operator is allowed to touch or be surrounded by
398
+ # arbitrary whitespace so forms like `flag='t'` and `flag <> 'f'` all
399
+ # collapse to the same canonical shape. Inequality is preserved as `!=`
400
+ # because `flag <> 't'` is not the same as `flag = 'f'` (NULL handling
401
+ # differs), so they must not share a canonical form. The lookbehind holds
402
+ # the equality patterns to a standalone `=`, so the ordering comparison in
403
+ # `note >= 't'` keeps both its operator and its value.
404
+ sql
405
+ .gsub(/(?<![<>!])\s*=\s*'t'/, ' = 1')
406
+ .gsub(/(?<![<>!])\s*=\s*'f'/, ' = 0')
407
+ .gsub(/\s*<>\s*'t'/, ' != 1')
408
+ .gsub(/\s*<>\s*'f'/, ' != 0')
409
+ .gsub(/\s*!=\s*'t'/, ' != 1')
410
+ .gsub(/\s*!=\s*'f'/, ' != 0')
411
+ end
412
+
413
+ # Rewrites the `TRUE` / `FALSE` / `NULL` keywords and the `IS` phrasings
414
+ # around them to one spelling. These run once literals are masked, so a
415
+ # value that happens to read `IS TRUE` keeps its own text.
416
+ def normalize_boolean_and_null_keywords(sql)
249
417
  normalized_sql = sql.dup
418
+ # `IS NOT TRUE` / `IS NOT FALSE` are matched before the bare `IS TRUE` /
419
+ # `IS FALSE` forms so the longer phrase wins. They normalize to `IS NOT 1`
420
+ # / `IS NOT 0` rather than `= 0` / `= 1` because `IS NOT TRUE` is not the
421
+ # same as `= FALSE` (NULL handling differs).
422
+ normalized_sql = normalized_sql.gsub(/\bIS\s+NOT\s+TRUE\b/i, ' IS NOT 1')
423
+ normalized_sql = normalized_sql.gsub(/\bIS\s+NOT\s+FALSE\b/i, ' IS NOT 0')
424
+ # `/\bIS\s+TRUE\b/i` and `/\bIS\s+FALSE\b/i` normalize predicate forms
425
+ # like `flag IS TRUE` to `flag = 1` so they match `flag = TRUE` and
426
+ # `flag = 't'`.
427
+ normalized_sql = normalized_sql.gsub(/\bIS\s+TRUE\b/i, ' = 1')
428
+ normalized_sql = normalized_sql.gsub(/\bIS\s+FALSE\b/i, ' = 0')
250
429
  # `/\bTRUE\b/i` and `/\bFALSE\b/i` normalize boolean literals to `1` / `0`
251
430
  # so they match SQL generated by Active Record on some adapters.
252
431
  normalized_sql = normalized_sql.gsub(/\bTRUE\b/i, '1').gsub(/\bFALSE\b/i, '0')
@@ -254,27 +433,88 @@ module DatabaseConsistency
254
433
  normalized_sql = normalized_sql.gsub(/\bIS\s+NOT\s+NULL\b/i, ' IS NOT NULL')
255
434
  # `/\bIS\s+NULL\b/i` normalizes `IS NULL` spacing and casing.
256
435
  normalized_sql = normalized_sql.gsub(/\bIS\s+NULL\b/i, ' IS NULL')
257
- # `/ = 't'/` and `/ = 'f'/` normalize PostgreSQL boolean literals stored
258
- # as `'t'` / `'f'` inside comparisons.
259
- normalized_sql = normalized_sql.gsub(/ = 't'/, ' = 1').gsub(/ = 'f'/, ' = 0')
260
436
  normalized_sql.gsub(/\s+/, ' ').strip
261
437
  end
262
438
 
263
- # Normalizations that run while string literals are masked.
264
- def normalize_sql_post_mask(sql)
439
+ # Rewrites exponent notation as the plain decimal PostgreSQL itself writes
440
+ # when it expands a literal, so `1e+20` and the `1.0e+20` Active Record
441
+ # generates reach the same string. The digits are shifted as text rather
442
+ # than through a float, so a wide value keeps every one of them.
443
+ def expand_exponent_literals(sql)
444
+ sql.gsub(EXPONENT_LITERAL) do
445
+ match = Regexp.last_match
446
+ shift_decimal_point(match[1], "#{match[2]}#{match[3]}", match[2].length + match[4].to_i)
447
+ end
448
+ end
449
+
450
+ # Places the decimal point `position` digits into `digits`, padding with
451
+ # zeros on whichever side falls short and dropping a fraction that ends in
452
+ # them, so `1e-20` and `1.0e-20` land on the same digits. A zero that only
453
+ # holds the decimal point's place goes too, so `0.1e+2` reaches `10`.
454
+ def shift_decimal_point(sign, digits, position)
455
+ expanded =
456
+ if position >= digits.length
457
+ digits + ('0' * (position - digits.length))
458
+ elsif position.positive?
459
+ "#{digits[0...position]}.#{digits[position..]}"
460
+ else
461
+ "0.#{'0' * -position}#{digits}"
462
+ end
463
+ # On Ruby < 3.0, frozen strings forbid `sub!`.
464
+ expanded = expanded.sub(/\A0+(?=\d)/, '')
465
+
466
+ "#{sign}#{expanded}".sub(/(\.\d*?)0+\z/, '\1').chomp('.')
467
+ end
468
+
469
+ # Gives every operator a space either side, every comma a space after it
470
+ # and no parenthesis a space on its inside, which is how PostgreSQL writes
471
+ # an indexdef however the index was typed.
472
+ def normalize_operator_spacing(sql)
473
+ spaced_sql = sql.gsub(MASKED_LITERAL_OR_OPERATOR) do |match|
474
+ match.match?(MASKED_LITERAL) ? match : " #{match.strip} "
475
+ end
476
+ spaced_sql = spaced_sql.gsub(/\s*,\s*/, ', ')
477
+ spaced_sql.gsub(/\(\s+/, '(').gsub(/\s+\)/, ')')
478
+ end
479
+
480
+ # Rewrites the spellings that differ between adapters, or between what an
481
+ # adapter stores and what Active Record writes: quoted identifiers, casts,
482
+ # exponent notation, the spacing of an `IN` list, of operators and of
483
+ # commas, the parentheses PostgreSQL adds around a cast operand and the `<>`
484
+ # it writes for inequality. Literals are masked throughout, so none of it
485
+ # reaches the inside of a value.
486
+ def normalize_adapter_syntax(sql)
265
487
  # Strips quoted identifiers (double quotes on PostgreSQL/SQLite,
266
488
  # backticks on MySQL) so the same column normalizes across adapters.
267
489
  normalized_sql = sql.gsub(/["`]/, '')
268
- # `/::\w+/` removes PostgreSQL casts like `column::text`.
269
- normalized_sql = normalized_sql.gsub(/::\w+/, '')
270
- # `/\(([a-z_][\w.]*)\)/i` unwraps a bare identifier surrounded by
271
- # parentheses, e.g. `(internal_name)` -> `internal_name`.
272
- normalized_sql = normalized_sql.gsub(/\(([a-z_][\w.]*)\)/i, '\1')
273
- # `/\s*<>\s*/` rewrites the SQL inequality operator `<>` to `!=`.
274
- normalized_sql = normalized_sql.gsub(/\s*<>\s*/, ' != ')
490
+ normalized_sql = normalized_sql.gsub(CONDITION_CAST, '')
491
+ normalized_sql = expand_exponent_literals(normalized_sql)
492
+ # Gives `IN` one space before its list, so `qty IN(1)` and `qty IN (1)`
493
+ # reach the same string and the list is recognisable to the unwrappers
494
+ # below. `\b` keeps a call such as `min(1)` out of it.
495
+ normalized_sql = normalized_sql.gsub(/\bIN\s*\(/i, 'IN (')
496
+ normalized_sql = normalize_operator_spacing(normalized_sql)
497
+ normalized_sql = unwrap_redundant_parentheses(normalized_sql)
498
+ # Rewrites the SQL inequality operator `<>` to `!=`; the spacing step
499
+ # has already given it a space either side.
500
+ normalized_sql = normalized_sql.gsub('<>', '!=')
275
501
  normalized_sql.gsub(/\s+/, ' ').strip
276
502
  end
277
503
 
504
+ # Removes the parentheses PostgreSQL puts around an operand it had to cast,
505
+ # which are redundant once the cast itself is gone: `(0)::numeric` -> `0`,
506
+ # `((name)::character varying(3))::text` -> `name`, `(abs(1))::numeric` ->
507
+ # `abs(1)`. Each pass repeats because removing one layer can expose another.
508
+ def unwrap_redundant_parentheses(sql)
509
+ normalized_sql = sql.dup
510
+
511
+ true while normalized_sql.gsub!(WRAPPED_IDENTIFIER, '\1')
512
+ true while normalized_sql.gsub!(WRAPPED_NUMBER, '\1')
513
+ true while normalized_sql.gsub!(WRAPPED_FUNCTION_CALL, '\k<call>')
514
+
515
+ normalized_sql
516
+ end
517
+
278
518
  # Repeatedly removes one wrapping layer of parentheses when the whole SQL
279
519
  # fragment is enclosed, e.g. `((foo))` -> `foo`.
280
520
  def strip_outer_parentheses(sql)
@@ -317,52 +557,51 @@ module DatabaseConsistency
317
557
  def normalize_boolean_predicates(sql)
318
558
  normalized_sql = sql.dup
319
559
 
320
- # Matches a bare negated boolean predicate such as `NOT archived`
321
- # appearing at the start of an expression, after `AND` / `OR`, or after
322
- # an opening parenthesis, and rewrites it to `archived = 0`.
323
- normalized_sql.gsub!(
324
- /(^|(?:\bAND\b|\bOR\b|\())\s*NOT\s+([a-z_][\w.]*)\s*(?=$|(?:\bAND\b|\bOR\b|\)))/i
325
- ) { "#{Regexp.last_match(1)} #{Regexp.last_match(2)} = 0" }
560
+ normalized_sql.gsub!(NEGATED_BOOLEAN_PREDICATE) do
561
+ "#{Regexp.last_match(1)} #{Regexp.last_match(2)} = 0"
562
+ end
326
563
 
327
- # Matches a bare boolean predicate such as `most_recent` appearing in the
328
- # same structural positions, and rewrites it to `most_recent = 1`.
329
- normalized_sql.gsub!(
330
- /(^|(?:\bAND\b|\bOR\b|\())\s*([a-z_][\w.]*)\s*(?=$|(?:\bAND\b|\bOR\b|\)))/i
331
- ) { "#{Regexp.last_match(1)} #{Regexp.last_match(2)} = 1" }
564
+ normalized_sql.gsub!(BARE_BOOLEAN_PREDICATE) do
565
+ "#{Regexp.last_match(1)} #{Regexp.last_match(2)} = 1"
566
+ end
332
567
 
333
568
  normalized_sql.gsub(/\s+/, ' ').strip
334
569
  end
335
570
 
336
- # Rewrites PostgreSQL's `= ANY (ARRAY[...])` form into an `IN (...)` form
337
- # so it matches the SQL Active Record typically generates for arrays.
571
+ # Rewrites PostgreSQL's `= ANY (ARRAY[...])` and `<> ALL (ARRAY[...])` forms
572
+ # into the `IN (...)` and `NOT IN (...)` Active Record generates for arrays.
573
+ # `<>` has already become `!=` by this point in the pipeline.
338
574
  def normalize_array_any_predicates(sql)
339
- sql.gsub(
340
- # Matches `column = ANY (ARRAY[...])`, capturing the column name and the
341
- # full array payload so it can be converted to `column IN (...)`.
342
- /([a-z_][\w.]*)\s*=\s*ANY\s*\(ARRAY\[(.*?)\]\)/i
343
- ) { "#{Regexp.last_match(1)} IN (#{Regexp.last_match(2).gsub(/\s+/, ' ').strip})" }
575
+ sql.gsub(ARRAY_MEMBERSHIP_PREDICATE) do
576
+ match = Regexp.last_match
577
+ membership = match[:operator].match?(/ANY/i) ? 'IN' : 'NOT IN'
578
+
579
+ "#{match[:column]} #{membership} (#{match[:items].gsub(/\s+/, ' ').strip})"
580
+ end
344
581
  end
345
582
 
346
583
  # Rewrites negated "blank or nil" predicates into the same shape used by
347
584
  # `allow_blank`-derived guards: `IS NOT NULL AND != ''`.
348
585
  def normalize_negated_blank_or_nil_predicates(sql)
349
- sql.gsub(
350
- # Matches SQL like `NOT (column = '' OR column IS NULL)` while enforcing
351
- # the same column name on both sides via backreference `\1`.
352
- /NOT\s+\(\s*\(?([a-z_][\w.]*)\s*=\s*''\s+OR\s+\1\s+IS\s+NULL\)?\s*\)/i
353
- ) { "#{Regexp.last_match(1)} IS NOT NULL AND #{Regexp.last_match(1)} != ''" }
586
+ sql.gsub(NEGATED_BLANK_OR_NIL_PREDICATE) do
587
+ "#{Regexp.last_match(1)} IS NOT NULL AND #{Regexp.last_match(1)} != ''"
588
+ end
354
589
  end
355
590
 
356
591
  # Sorts simple `AND` clauses so `a AND b` and `b AND a` normalize to the
357
- # same string before comparison.
358
- def sort_and_clauses(sql)
592
+ # same string before comparison. Two clauses can be identical apart from the
593
+ # string each one compares against, and then those strings decide the order,
594
+ # which is why the literals go back in before the sort. A placeholder is
595
+ # numbered by where its literal appeared, so sorting on the placeholders
596
+ # would leave such a pair in whichever order it arrived in.
597
+ def sort_and_clauses(sql, literals)
359
598
  # Matches `AND` with surrounding whitespace and splits the expression into
360
599
  # comparable clause fragments.
361
600
  clauses = sql.split(/\s+AND\s+/i)
362
601
  return sql if clauses.length == 1
363
602
 
364
603
  clauses.map! { |clause| strip_outer_parentheses(clause) }
365
- clauses.sort.join(' AND ')
604
+ clauses.sort_by { |clause| unmask_condition_literals(clause, literals) }.join(' AND ')
366
605
  end
367
606
 
368
607
  # Builds the implicit SQL guard introduced by validator options that skip
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module DatabaseConsistency
4
- VERSION = '3.0.12'
4
+ VERSION = '3.0.14'
5
5
  end
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: database_consistency
3
3
  version: !ruby/object:Gem::Version
4
- version: 3.0.12
4
+ version: 3.0.14
5
5
  platform: ruby
6
6
  authors:
7
7
  - Evgeniy Demin
8
8
  autorequire:
9
9
  bindir: bin
10
10
  cert_chain: []
11
- date: 2026-09-15 00:00:00.000000000 Z
11
+ date: 2026-09-30 00:00:00.000000000 Z
12
12
  dependencies:
13
13
  - !ruby/object:Gem::Dependency
14
14
  name: activerecord