schema_reaper 1.0.13 → 1.0.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 920a3c80a9b99271cfb1152f9a4682dd42c03516934fbd6971c56ef0af06882f
4
- data.tar.gz: 63567586bef79cb15f27be238def5882fe3119955bb45bea346b4f979599e8ff
3
+ metadata.gz: 6f169920afef1838471a418e29ad207b6c9d3193ef57caf6678cf68d675a3405
4
+ data.tar.gz: 3e5d507411e727278004def7a752db662620a01371625ce761af4495fd020f2e
5
5
  SHA512:
6
- metadata.gz: fa80aecb2dacdbf843f03734818d739b7d5d386ba4096fc69163dacbc55ef52fb92a1b93adc1c4f5f4bf14bb5a1dc3a449fab647101d5a4b43a9922fa7088469
7
- data.tar.gz: 410917c5ece25227dbcfeb0b32128b8d0105770631c10a4175da273c3c36744b4bed428857db38b2b621fb4d5ffd03fce89e70d4599bae9c66676371ce42687f
6
+ metadata.gz: 739b0cf29def40063c3a182ad5b9d207f7409f9277e134e76bd980bc3db276a333f225a1974bf5c30a3e14af769d0c8128efeb46961d084f1349fc17bddd6eca
7
+ data.tar.gz: 692ed7533e2553ebbc9085001f1d805519149fe988860e23da9902b941de60c292a8918e60f2fe9c2d0e558aaeacd182d31e6376570e7ea9e8551977d5856db8
@@ -3,7 +3,8 @@
3
3
  module SchemaReaper
4
4
  module Analyzers
5
5
  # An index whose column list is a leading prefix of another index on the
6
- # same table is redundant (the wider index serves both).
6
+ # same table is redundant (the wider index serves both), and two indexes
7
+ # with identical column lists are redundant with each other.
7
8
  class DuplicateIndex < Base
8
9
  Registry.register(self)
9
10
 
@@ -14,28 +15,79 @@ module SchemaReaper
14
15
 
15
16
  private
16
17
 
18
+ # A partial index only exists for rows matching its WHERE clause and is
19
+ # usually there on purpose (a smaller, faster index for one condition),
20
+ # so it is excluded entirely rather than reasoned about: not a
21
+ # candidate for removal, and not a stand-in for a full index either.
22
+ # Comparing WHERE clauses for implication is out of scope here, and a
23
+ # wrong guess in either direction is a real index gone from production.
17
24
  def dupes_in(table)
18
- non_pk = table.indexes.reject(&:primary)
19
- non_pk.filter_map do |ix|
20
- covering = non_pk.find { |o| o != ix && !o.unique && o.covers?(ix) && o.columns != ix.columns }
21
- next unless covering
22
-
23
- finding(
24
- type: :duplicate_index,
25
- table: table.name,
26
- index: ix.name,
27
- column: ix.columns.join(","),
28
- severity: :low,
29
- confidence: 0.8,
30
- bytes_per_row: 0,
31
- evidence: [
32
- "#{ix.name} (#{ix.columns.join(", ")}) is a prefix of " \
33
- "#{covering.name} (#{covering.columns.join(", ")})"
34
- ],
35
- suggested_fix: "remove_index :#{table.name}, name: :#{ix.name}"
36
- )
25
+ non_pk = table.indexes.reject { |index| index.primary || index.partial? }
26
+ by_columns = non_pk.group_by(&:columns)
27
+ non_pk.filter_map { |index| finding_for(table, index, non_pk, by_columns) }
28
+ end
29
+
30
+ def finding_for(table, index, non_pk, by_columns)
31
+ covering = covering_for(index, non_pk, by_columns)
32
+ return unless covering
33
+
34
+ finding(
35
+ type: :duplicate_index,
36
+ table: table.name,
37
+ index: index.name,
38
+ column: index.columns.join(","),
39
+ severity: :low,
40
+ confidence: 0.8,
41
+ bytes_per_row: 0,
42
+ evidence: [evidence_for(index, covering)],
43
+ suggested_fix: "remove_index :#{table.name}, name: :#{index.name}"
44
+ )
45
+ end
46
+
47
+ # Two indexes with the same column list have no natural wider/narrower
48
+ # direction, so comparing them pairwise would have each flag the other --
49
+ # applying both suggested fixes would then drop the column pair
50
+ # entirely. Instead, one designated survivor per exact-column group is
51
+ # chosen once; every other member of the group is redundant against it,
52
+ # and the survivor itself is only checked against a genuinely wider
53
+ # prefix elsewhere on the table.
54
+ def covering_for(index, non_pk, by_columns)
55
+ peers = by_columns[index.columns]
56
+ return prefix_covering_for(index, non_pk) if peers.size == 1
57
+
58
+ survivor = survivor_of(peers)
59
+ return survivor unless survivor == index
60
+
61
+ prefix_covering_for(index, non_pk)
62
+ end
63
+
64
+ # Keep a unique index over a non-unique one -- it enforces a guarantee
65
+ # the others don't -- breaking further ties by name for a stable,
66
+ # order-independent choice.
67
+ def survivor_of(peers)
68
+ unique_peers = peers.select(&:unique)
69
+ (unique_peers.empty? ? peers : unique_peers).min_by(&:name)
70
+ end
71
+
72
+ def prefix_covering_for(index, non_pk)
73
+ non_pk.find do |o|
74
+ o != index && o.columns != index.columns && o.covers?(index) && safe_to_drop?(index, against: o)
37
75
  end
38
76
  end
77
+
78
+ # A wider index does not make a narrower prefix unique: unique (a, b)
79
+ # says nothing about whether a alone is unique. So a unique index is
80
+ # only safe to drop in favour of another index that is itself unique on
81
+ # that exact same column list -- never a merely-wider or non-unique one.
82
+ def safe_to_drop?(index, against:)
83
+ !index.unique || (against.unique && against.columns == index.columns)
84
+ end
85
+
86
+ def evidence_for(index, covering)
87
+ relation = covering.columns == index.columns ? "duplicates" : "is a prefix of"
88
+ "#{index.name} (#{index.columns.join(", ")}) #{relation} " \
89
+ "#{covering.name} (#{covering.columns.join(", ")})"
90
+ end
39
91
  end
40
92
  end
41
93
  end
@@ -100,22 +100,16 @@ module SchemaReaper
100
100
  end
101
101
  end
102
102
 
103
- # Columns must come back in index-key order, not table order: prefix
104
- # comparisons (Index#covers?) are only meaningful on the real key order.
105
- #
106
- # pg_get_indexdef(indexrelid, column_no, pretty) is keyed by column
107
- # *position* (1..indnkeyatts), not by table attnum, so it renders a
108
- # plain column and an expression column uniformly. The previous query
109
- # joined each indkey entry against pg_attribute by attnum; an expression
110
- # column's indkey entry is 0, which matches no real column, so the join
111
- # silently dropped it -- an index on (COALESCE(a, b), c) came back as
112
- # just "c", which then looked like a genuine duplicate of any ordinary
113
- # index on :c. indnkeyatts also excludes INCLUDE columns, which are
114
- # payload only and never participate in Index#covers?'s prefix check.
103
+ # Columns come back in index-key order via pg_get_indexdef(indexrelid,
104
+ # column_no, pretty), keyed by column position rather than table attnum
105
+ # so it renders a plain column and an expression uniformly and never
106
+ # silently drops one. indnkeyatts excludes INCLUDE columns, which are
107
+ # payload only. partial (indpred IS NOT NULL) lets analyzers refuse to
108
+ # treat a conditional index as if it covered every row.
115
109
  def indexes_for(table)
116
110
  exec(<<~SQL, [table]).map do |r|
117
111
  SELECT i.relname AS name, ix.indisunique AS "unique", ix.indisprimary AS "primary",
118
- s.idx_scan AS scans,
112
+ (ix.indpred IS NOT NULL) AS partial, s.idx_scan AS scans,
119
113
  array_to_string(
120
114
  array_agg(pg_get_indexdef(ix.indexrelid, k.ord::int, true) ORDER BY k.ord),
121
115
  '#{COLUMN_SEPARATOR}'
@@ -126,11 +120,11 @@ module SchemaReaper
126
120
  JOIN LATERAL generate_series(1, ix.indnkeyatts) AS k(ord) ON TRUE
127
121
  LEFT JOIN pg_stat_user_indexes s ON s.indexrelid = i.oid
128
122
  WHERE t.relname = $1
129
- GROUP BY i.relname, ix.indisunique, ix.indisprimary, s.idx_scan
123
+ GROUP BY i.relname, ix.indisunique, ix.indisprimary, ix.indpred, s.idx_scan
130
124
  SQL
131
125
  Index.new(
132
126
  name: r["name"], columns: r["cols"].split(COLUMN_SEPARATOR),
133
- unique: r["unique"] == "t", primary: r["primary"] == "t",
127
+ unique: r["unique"] == "t", primary: r["primary"] == "t", partial: r["partial"] == "t",
134
128
  scans: r["scans"]&.to_i
135
129
  )
136
130
  end
@@ -19,10 +19,18 @@ module SchemaReaper
19
19
  end
20
20
  end
21
21
 
22
- Index = Struct.new(:name, :columns, :unique, :primary, :scans, keyword_init: true) do
22
+ Index = Struct.new(:name, :columns, :unique, :primary, :scans, :partial, keyword_init: true) do
23
23
  def covers?(other)
24
24
  columns.first(other.columns.length) == other.columns
25
25
  end
26
+
27
+ # A partial index only exists for rows matching its WHERE clause, so it
28
+ # cannot stand in for a full index for rows outside that condition.
29
+ # comparing WHERE clauses for implication is out of scope here, so a
30
+ # partial index is simply never treated as covering another.
31
+ def partial?
32
+ !!partial
33
+ end
26
34
  end
27
35
 
28
36
  Table = Struct.new(
@@ -12,6 +12,16 @@ module SchemaReaper
12
12
  RUBY_GLOB = "**/*.rb"
13
13
  WORD_RE = /[a-z_][a-z0-9_]*/i.freeze
14
14
 
15
+ # Macros that generate a real column from a differently-named virtual
16
+ # attribute, so the literal column name never appears in application
17
+ # code. Missing this made dead_column flag has_secure_password's
18
+ # password_digest, and attr_encrypted/Lockbox/KMS ciphertext columns,
19
+ # as unused even though they're the live backing store.
20
+ DIGEST_MACROS = %w[has_secure_password].freeze
21
+ DIGEST_SUFFIXES = %w[_digest].freeze
22
+ ENCRYPTED_MACROS = %w[encrypts attr_encrypted lockbox_encrypts].freeze
23
+ ENCRYPTED_SUFFIXES = %w[_ciphertext _iv _tag _encrypted].freeze
24
+
15
25
  # Node classes whose #name (or #unescaped) is a bare identifier we treat
16
26
  # as a possible column/table reference.
17
27
  NAME_NODES = [
@@ -61,6 +71,7 @@ module SchemaReaper
61
71
  return unless node.is_a?(Prism::Node)
62
72
 
63
73
  out.merge(tokens_for(node))
74
+ out.merge(macro_derived_tokens(node)) if node.is_a?(Prism::CallNode)
64
75
  node.compact_child_nodes.each { |c| collect_from_node(c, out) }
65
76
  end
66
77
 
@@ -78,6 +89,34 @@ module SchemaReaper
78
89
  end
79
90
  end
80
91
 
92
+ # `has_secure_password` / `encrypts :field` / `attr_encrypted :field`
93
+ # never write their generated column name (password_digest,
94
+ # field_ciphertext, ...) anywhere in source -- only the virtual
95
+ # attribute name. Derive the column names a macro call implies so they
96
+ # count as "used" instead of looking dead.
97
+ def macro_derived_tokens(node)
98
+ call_name = node.name&.to_s
99
+ return [] unless call_name
100
+
101
+ if DIGEST_MACROS.include?(call_name)
102
+ attrs = macro_symbol_args(node)
103
+ attrs = ["password"] if attrs.empty?
104
+ attrs.flat_map { |a| DIGEST_SUFFIXES.map { |s| "#{a}#{s}" } }
105
+ elsif ENCRYPTED_MACROS.include?(call_name)
106
+ macro_symbol_args(node).flat_map { |a| ENCRYPTED_SUFFIXES.map { |s| "#{a}#{s}" } }
107
+ else
108
+ []
109
+ end
110
+ end
111
+
112
+ # Leading bare symbol arguments of a call, e.g. `encrypts :a, :b, purpose: :x`
113
+ # => ["a", "b"]. Stops at the first non-symbol (keyword args, etc).
114
+ def macro_symbol_args(node)
115
+ args = node.arguments&.arguments || []
116
+ args.take_while { |a| a.is_a?(Prism::SymbolNode) }
117
+ .map { |a| a.unescaped.to_s.downcase }
118
+ end
119
+
81
120
  def text_tokens(path)
82
121
  File.read(path).scan(WORD_RE).to_set(&:downcase)
83
122
  rescue StandardError
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module SchemaReaper
4
- VERSION = "1.0.13"
4
+ VERSION = "1.0.14"
5
5
  end
metadata CHANGED
@@ -1,14 +1,15 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: schema_reaper
3
3
  version: !ruby/object:Gem::Version
4
- version: 1.0.13
4
+ version: 1.0.14
5
5
  platform: ruby
6
6
  authors:
7
7
  - aksshatt
8
8
  - mitkush
9
+ autorequire:
9
10
  bindir: exe
10
11
  cert_chain: []
11
- date: 1980-01-02 00:00:00.000000000 Z
12
+ date: 2026-09-18 00:00:00.000000000 Z
12
13
  dependencies:
13
14
  - !ruby/object:Gem::Dependency
14
15
  name: prism
@@ -156,6 +157,7 @@ metadata:
156
157
  wiki_uri: https://github.com/aksshatt/schema_reaper/wiki
157
158
  funding_uri: https://github.com/sponsors/aksshatt
158
159
  rubygems_mfa_required: 'true'
160
+ post_install_message:
159
161
  rdoc_options: []
160
162
  require_paths:
161
163
  - lib
@@ -170,7 +172,8 @@ required_rubygems_version: !ruby/object:Gem::Requirement
170
172
  - !ruby/object:Gem::Version
171
173
  version: '0'
172
174
  requirements: []
173
- rubygems_version: 3.6.9
175
+ rubygems_version: 3.4.10
176
+ signing_key:
174
177
  specification_version: 4
175
178
  summary: Find and safely remove schema dead-weight in Rails + PostgreSQL apps.
176
179
  test_files: []