exhale 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,105 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "digest"
4
+ require "set"
5
+ require "bigdecimal"
6
+ require "bigdecimal/math"
7
+
8
+ module Exhale
9
+ module Dry
10
+ # A normalized node with its structural digest. Equal structure gives an
11
+ # equal digest, computed bottom up from the node's kind, label and its
12
+ # children's digests, so nothing gets printed.
13
+ class FNode
14
+ attr_reader :digest, :size, :start_line, :end_line, :children
15
+
16
+ def initialize(digest, size, start_line, end_line, children, sequence)
17
+ @digest = digest
18
+ @size = size
19
+ @start_line = start_line
20
+ @end_line = end_line
21
+ @children = children
22
+ @sequence = sequence
23
+ end
24
+
25
+ def sequence?
26
+ @sequence
27
+ end
28
+
29
+ def lines
30
+ end_line - start_line + 1
31
+ end
32
+
33
+ # Preorder, so a parent always comes before its descendants.
34
+ def each_node(&block)
35
+ return enum_for(:each_node) unless block
36
+
37
+ stack = [self]
38
+ until stack.empty?
39
+ node = stack.pop
40
+ yield node
41
+ node.children.reverse_each { |child| stack.push(child) }
42
+ end
43
+ end
44
+
45
+ # The fingerprint set: the digest of every subtree.
46
+ def digests
47
+ set = Set.new
48
+ each_node { |node| set << node.digest }
49
+ set
50
+ end
51
+ end
52
+
53
+ module Fingerprints
54
+ module_function
55
+
56
+ # Digests are the first 8 bytes of SHA-256, which is unseeded and the
57
+ # same in every process. Ruby's own String#hash is seeded per process,
58
+ # so it would break the on-disk cache and determinism both.
59
+ def build(shape)
60
+ children = shape.children.map { |child| build(child) }
61
+ # Bytes, not text: a Unicode label's UTF-8 next to packed digests
62
+ # would be two incompatible encodings. ASCII digests are unchanged.
63
+ payload = "#{shape.kind}\u0000#{shape.label}\u0000".b
64
+ children.each { |child| payload << [child.digest].pack("Q>") }
65
+ digest = Digest::SHA256.digest(payload).unpack1("Q>")
66
+ size = 1 + children.sum(&:size)
67
+ FNode.new(digest, size, shape.start_line, shape.end_line, children, shape.sequence ? true : false)
68
+ end
69
+
70
+ # The digest of a run of sibling nodes, used for statement-run fragments.
71
+ def run_digest(nodes)
72
+ Digest::SHA256.digest(nodes.map(&:digest).pack("Q>*")).unpack1("Q>")
73
+ end
74
+ end
75
+
76
+ # w(f) = ln(1 + C / df(f)), stored as fixed-point integers.
77
+ #
78
+ # A weight depends only on its own fingerprint's count, never on the size
79
+ # of the codebase, which is what keeps an incremental sweep exact. The log
80
+ # comes from BigDecimal rather than Math.log: libm can differ in the last
81
+ # bit between platforms, and BigDecimal's arithmetic is the same
82
+ # everywhere.
83
+ class Weights
84
+ C = 1000
85
+ SCALE = 1_000_000
86
+ PRECISION = 30
87
+
88
+ def initialize(counts)
89
+ @counts = counts
90
+ @by_count = {}
91
+ end
92
+
93
+ def [](digest)
94
+ for_count(@counts.fetch(digest, 1))
95
+ end
96
+
97
+ def for_count(count)
98
+ @by_count[count] ||= begin
99
+ ratio = BigDecimal(1) + BigDecimal(C).div(count, PRECISION)
100
+ (BigMath.log(ratio, PRECISION) * SCALE).round
101
+ end
102
+ end
103
+ end
104
+ end
105
+ end
@@ -0,0 +1,425 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "set"
4
+ require_relative "../errors"
5
+
6
+ module Exhale
7
+ module Dry
8
+ # One cluster of duplicated code in the report. copy is the newest
9
+ # location; others are the rest, each with its score against the copy.
10
+ Finding = Struct.new(:klass, :kind, :score, :copy, :others, :hint, :payoff, :primitive, :touched,
11
+ keyword_init: true)
12
+ Kept = Struct.new(:score, :a, :b, :clause, :new_clause, keyword_init: true)
13
+ # a and b are the two sides' identities; a pair is contracted by its
14
+ # keys, but reported by the names a reader knows.
15
+ Contracted = Struct.new(:a, :b, :score, keyword_init: true)
16
+ Result = Struct.new(:findings, :kept, :contracted, :clause_errors, :parse_errors, :base_sha, :exit_code,
17
+ :notes, :introduced_only, :touched, keyword_init: true)
18
+
19
+ # What a base sweep leaves behind for labeling: which finding each
20
+ # location sat in there, found by its key or by its structure, every
21
+ # pair that matched, and which clauses existed.
22
+ #
23
+ # clusters - {location key => cluster id}
24
+ # structures - {structural key => [cluster id, ...]}
25
+ # pairs - [[key a, key b, score, structural a, structural b, identity a, identity b], ...]
26
+ # clause_keys - Set of clause keys
27
+ BaseSummary = Struct.new(:clusters, :structures, :pairs, :clause_keys, keyword_init: true)
28
+
29
+ # Turns a sweep's matches into the verdict. The verdict reads only the
30
+ # tree being checked; the base summary only labels findings, except in
31
+ # the introduced-only on-ramp.
32
+ class Gate
33
+ ORDER = { introduced: 0, shifted: 1, already_there: 2, found: 3 }.freeze
34
+ FAILS_ON_RAMP = %i[introduced shifted found].freeze
35
+
36
+ # Every base match, kept or not, joins its two locations into one base
37
+ # cluster, the same union the head's findings come from. A label then
38
+ # asks whether two locations sat in one cluster, so a relationship the
39
+ # matcher only implied (A~B and B~C, or a star around a hub) still
40
+ # counts.
41
+ def self.summarize(sweep)
42
+ sets = UnionFind.new
43
+ sweep.matches.each { |m| sets.union(m.a.key, m.b.key) }
44
+ roots = sets.keys.group_by { |key| sets.find(key) }.values.map(&:sort).sort
45
+ clusters = roots.each_with_index.each_with_object({}) { |(keys, id), map| keys.each { |key| map[key] = id } }
46
+ structures = Hash.new { |hash, key| hash[key] = Set.new }
47
+ sweep.matches.each do |m|
48
+ [m.a, m.b].each { |l| structures[l.structural_key] << clusters.fetch(l.key) }
49
+ end
50
+
51
+ BaseSummary.new(
52
+ clusters: clusters,
53
+ structures: structures.transform_values { |ids| ids.to_a.sort },
54
+ pairs: sweep.matches.map { |m| pair_row(m) }.sort_by { |row| row.values_at(0, 1, 3, 4) },
55
+ clause_keys: Set.new(sweep.contract.clauses.map(&:key))
56
+ )
57
+ end
58
+
59
+ def self.pair_row(match)
60
+ a, b = [match.a, match.b].sort_by(&:key)
61
+ [a.key, b.key, match.score, *[a.structural_key, b.structural_key].sort, a.unit.identity, b.unit.identity]
62
+ end
63
+
64
+ # Disjoint sets over any hashable keys.
65
+ class UnionFind
66
+ def initialize
67
+ @parent = {}
68
+ end
69
+
70
+ def keys
71
+ @parent.keys
72
+ end
73
+
74
+ def find(key)
75
+ @parent[key] = key unless @parent.key?(key)
76
+ root = key
77
+ root = @parent[root] until @parent[root] == root
78
+ while @parent[key] != root
79
+ @parent[key], key = root, @parent[key]
80
+ end
81
+ root
82
+ end
83
+
84
+ def union(a, b)
85
+ root_a = find(a)
86
+ root_b = find(b)
87
+ @parent[root_b] = root_a unless root_a == root_b
88
+ end
89
+ end
90
+
91
+ def initialize(head:, base:, base_sha:, git:, changed_lines:, introduced_only:, paths:, overrides:)
92
+ @head = head
93
+ @base = base
94
+ @base_sha = base_sha
95
+ @git = git
96
+ @changed_lines = changed_lines
97
+ @introduced_only = introduced_only
98
+ @paths = paths
99
+ @overrides = overrides
100
+ @blame = {}
101
+ end
102
+
103
+ def verdict
104
+ kept, unkept, used = partition
105
+ clusters = cluster(unkept)
106
+ prefetch_blame(clusters)
107
+ findings = clusters.map { |locations, matches| finding(locations, matches) }
108
+ findings = narrow(findings) unless @paths.empty?
109
+ findings.sort_by! do |f|
110
+ [ORDER.fetch(f.klass), -f.payoff, f.copy.path, f.copy.start_line, f.copy.end_line, f.copy.unit.identity]
111
+ end
112
+ errors = clause_errors(used)
113
+
114
+ Result.new(findings: findings, kept: kept, contracted: contracted, clause_errors: errors,
115
+ parse_errors: @head.parse_errors, base_sha: @base_sha, notes: notes,
116
+ introduced_only: @introduced_only, touched: @changed_lines.size,
117
+ exit_code: exit_code(findings, errors))
118
+ end
119
+
120
+ private
121
+
122
+ def partition
123
+ used = Set.new
124
+ kept = []
125
+ unkept = []
126
+ @head.matches.each do |match|
127
+ clause = @head.resolver.keeping_clause(match.a.unit, match.b.unit)
128
+ if clause
129
+ used << clause.key
130
+ kept << Kept.new(score: match.score, a: match.a, b: match.b, clause: clause,
131
+ new_clause: @base ? !@base.clause_keys.include?(clause.key) : false)
132
+ else
133
+ unkept << match
134
+ end
135
+ end
136
+ [kept, unkept + unkept_twins(kept), used]
137
+ end
138
+
139
+ # The matcher joins a group of identical units as a star around one
140
+ # hub, so a kept edge from the hub says nothing about the members the
141
+ # star never paired. Each kept edge between whole units stands for
142
+ # every pair across the two groups it touches (or within the one group),
143
+ # and each of those pairs the Contract doesn't keep is unkept
144
+ # duplication. A group no clause touches needs no expansion: its star
145
+ # edges are unkept already.
146
+ def unkept_twins(kept)
147
+ groups = @head.matches.flat_map { |m| [m.a, m.b] }.select(&:whole).uniq(&:id)
148
+ .group_by { |l| l.entry.tree.digest }
149
+ emitted = Set.new(@head.matches.map { |m| [m.a.id, m.b.id].sort })
150
+ expanded = Set.new
151
+ kept.each_with_object([]) do |k, extra|
152
+ next unless k.a.whole && k.b.whole
153
+
154
+ pair = [k.a.entry.tree.digest, k.b.entry.tree.digest].sort
155
+ next unless expanded.add?(pair)
156
+
157
+ left = groups.fetch(pair[0])
158
+ right = groups.fetch(pair[1])
159
+ next if left.size == 1 && right.size == 1
160
+
161
+ candidates = pair[0] == pair[1] ? left.combination(2) : left.product(right)
162
+ candidates.each do |a, b|
163
+ next if emitted.include?([a.id, b.id].sort)
164
+ next if @head.resolver.keeping_clause(a.unit, b.unit)
165
+
166
+ match = twin_match(a, b, pair[0] == pair[1] ? Rational(1) : k.score)
167
+ extra << match if match
168
+ end
169
+ end
170
+ end
171
+
172
+ # Identical units share their fingerprints, so a pair across two groups
173
+ # scores what their hubs scored. Each pair still meets its own settings.
174
+ def twin_match(a, b, score)
175
+ settings = @head.settings_for_pair(a.unit, b.unit)
176
+ return unless [a, b].all? { |l| l.lines >= settings[:min_lines] && l.size >= settings[:min_nodes] }
177
+ return if score < settings[:threshold]
178
+
179
+ Match.new(a: a, b: b, score: score, kind: :unit)
180
+ end
181
+
182
+ # Matches that share a location merge, so one shape copied into four
183
+ # controllers is one finding with four locations.
184
+ def cluster(matches)
185
+ parent = {}
186
+ find = lambda do |id|
187
+ parent[id] = id unless parent.key?(id)
188
+ root = id
189
+ root = parent[root] until parent[root] == root
190
+ parent[id] = root
191
+ end
192
+
193
+ location_of = {}
194
+ matches.each do |match|
195
+ a = location_id(match.a)
196
+ b = location_id(match.b)
197
+ location_of[a] = match.a
198
+ location_of[b] = match.b
199
+ root_a = find.call(a)
200
+ root_b = find.call(b)
201
+ parent[root_b] = root_a unless root_a == root_b
202
+ end
203
+
204
+ groups = Hash.new { |hash, root| hash[root] = [[], []] }
205
+ location_of.each_key { |id| groups[find.call(id)][0] << location_of[id] }
206
+ matches.each { |match| groups[find.call(location_id(match.a))][1] << match }
207
+ groups.values
208
+ end
209
+
210
+ def location_id(location)
211
+ [location.entry.id, location.start_line, location.end_line]
212
+ end
213
+
214
+ def finding(locations, matches)
215
+ @scores = matches.to_h { |m| [[location_id(m.a), location_id(m.b)].sort, m.score] }
216
+ ordered = locations.sort_by { |l| [age(l), l.path, l.start_line, l.end_line] }
217
+ original = ordered.first
218
+ copy = ordered.last
219
+ others = (locations - [copy]).map { |l| [l, score_between(copy, l)] }
220
+ others.sort_by! { |l, score| [-score, l.path, l.start_line, l.end_line] }
221
+ touched = locations.select { |l| touched?(l) }
222
+ primitive = @head.resolver.primitive_for(original.unit)&.name
223
+
224
+ Finding.new(klass: label(copy, original, locations, touched), kind: kind_of(locations),
225
+ score: others.first[1], copy: copy, others: others, primitive: primitive, touched: touched,
226
+ hint: hint(copy, others, locations, touched, primitive),
227
+ payoff: payoff(locations, original, copy))
228
+ end
229
+
230
+ BLAME_THREADS = 8
231
+
232
+ # One blame per file, run a few at a time. Each result depends only on
233
+ # its own file, so the order the threads finish in changes nothing.
234
+ def prefetch_blame(clusters)
235
+ return unless @git
236
+
237
+ paths = clusters.flat_map { |locations, _| locations.reject { |l| touched?(l) }.map(&:path) }.uniq
238
+ queue = Queue.new
239
+ paths.each { |path| queue << path }
240
+ queue.close
241
+ results = {}
242
+ lock = Mutex.new
243
+ Array.new([BLAME_THREADS, paths.size].min) do
244
+ Thread.new do
245
+ while (path = queue.pop)
246
+ times = begin
247
+ @git.blame_times(path)
248
+ rescue GitError
249
+ []
250
+ end
251
+ lock.synchronize { results[path] = times }
252
+ end
253
+ end
254
+ end.each(&:join)
255
+ @blame.merge!(results)
256
+ end
257
+
258
+ # Older lines are the original. A touched location is the newest thing
259
+ # there is; with no git, path order decides.
260
+ def age(location)
261
+ return Float::INFINITY if touched?(location)
262
+ return 0 unless @git
263
+
264
+ times = (@blame[location.path] ||= @git.blame_times(location.path))
265
+ times[(location.start_line - 1)..(location.end_line - 1)].compact.max || 0
266
+ rescue GitError
267
+ 0
268
+ end
269
+
270
+ def touched?(location)
271
+ lines = @changed_lines[location.path]
272
+ lines && (location.start_line..location.end_line).any? { |line| lines.include?(line) }
273
+ end
274
+
275
+ # A hash lookup per pair, so a finding with hundreds of copies stays
276
+ # cheap; only pairs the matcher never scored directly get computed.
277
+ def score_between(a, b)
278
+ @scores.fetch([location_id(a), location_id(b)].sort) do
279
+ @head.index.score(a.set, total(a), b.set, total(b))
280
+ end
281
+ end
282
+
283
+ def total(location)
284
+ location.whole ? location.entry.total : @head.index.weight_of(location.set)
285
+ end
286
+
287
+ # Each touched location answers for itself: one that sat in no base
288
+ # finding with any other location in this finding is new, whatever
289
+ # else in the finding was there before. That way a PR can't hide a new
290
+ # copy behind an old pair by also editing an old copy.
291
+ #
292
+ # Pure moves touch no lines, so structure only vouches for a finding
293
+ # the PR left alone entirely: it's already there when every location
294
+ # sat in one base finding, each found by its key or by its structure.
295
+ def label(_copy, _original, locations, touched)
296
+ return :found unless @base
297
+
298
+ unless touched.empty?
299
+ introduced = touched.any? { |l| locations.none? { |other| !other.equal?(l) && base_pair?(l, other) } }
300
+ return introduced ? :introduced : :already_there
301
+ end
302
+ shared = locations.map { |l| base_clusters(l) }.reduce(:&)
303
+ shared.empty? ? :shifted : :already_there
304
+ end
305
+
306
+ def base_pair?(a, b)
307
+ cluster = @base.clusters[a.key]
308
+ !cluster.nil? && cluster == @base.clusters[b.key]
309
+ end
310
+
311
+ def base_clusters(location)
312
+ ids = Set.new(@base.structures.fetch(location.structural_key, []))
313
+ cluster = @base.clusters[location.key]
314
+ ids << cluster if cluster
315
+ ids
316
+ end
317
+
318
+ def kind_of(locations)
319
+ return locations.first.unit.kind if locations.all?(&:whole) && locations.map { |l| l.unit.kind }.uniq.size == 1
320
+
321
+ :fragment
322
+ end
323
+
324
+ def hint(copy, others, locations, touched, primitive)
325
+ text = base_hint(copy, others, locations)
326
+ text = "promote one copy to a shared primitive before merge" if touched.size == locations.size && @base
327
+ primitive ? "#{text}; it belongs to the #{primitive} primitive" : text
328
+ end
329
+
330
+ def base_hint(copy, others, locations)
331
+ existing = others.first[0]
332
+ if copy.unit.language == :erb
333
+ return "render the existing partial #{existing.unit.identity}" if existing.whole && partial?(existing)
334
+
335
+ return "extract a partial or a component"
336
+ end
337
+ return "extend or call #{existing.unit.identity}" if locations.all?(&:whole)
338
+
339
+ namespaces = locations.map { |l| l.unit.namespace }.uniq
340
+ return "extract a method" if namespaces.size == 1
341
+
342
+ where = "#{locations.size} copies in #{namespaces.size} classes"
343
+ if locations.all? { |l| l.path.start_with?("app/models/", "app/controllers/") }
344
+ "#{where}; extract a concern"
345
+ else
346
+ "#{where}; extract a concern, or a service object"
347
+ end
348
+ end
349
+
350
+ def partial?(location)
351
+ File.basename(location.path).start_with?("_")
352
+ end
353
+
354
+ # The code that would disappear if the cluster folded into one copy.
355
+ def payoff(locations, original, copy)
356
+ (locations - [original]).sum do |l|
357
+ score = l.equal?(copy) ? score_between(copy, original) : score_between(copy, l)
358
+ l.size * score
359
+ end.to_f.round(1)
360
+ end
361
+
362
+ # Paths arrive relative to the root and already checked to exist; "."
363
+ # is the whole tree.
364
+ def narrow(findings)
365
+ findings.select do |f|
366
+ [f.copy, *f.others.map(&:first)].any? do |l|
367
+ @paths.any? { |p| p == "." || l.path == p || l.path.start_with?("#{p}/") }
368
+ end
369
+ end
370
+ end
371
+
372
+ def clause_errors(used)
373
+ stale = @head.contract.clauses.reject { |clause| used.include?(clause.key) }.map do |clause|
374
+ ContractError.new(clause.path, clause.line,
375
+ "stale clause: nothing it covers duplicates anything over the threshold; delete it")
376
+ end
377
+ @head.contract.errors + @head.resolver.errors + stale
378
+ end
379
+
380
+ # Counted per occurrence: each structural pair the base matched more
381
+ # often than the head does lost that many pairs, so deleting one of
382
+ # three identical copies contracts one pair even though a pair of the
383
+ # same shape survives. A pair still matched under its own keys never
384
+ # counts as lost, and a pure move keeps its structure, so it doesn't
385
+ # either.
386
+ def contracted
387
+ return [] unless @base
388
+
389
+ keys = Set.new
390
+ structural = Hash.new(0)
391
+ @head.matches.each do |m|
392
+ keys << [m.a.key, m.b.key].sort
393
+ structural[[m.a.structural_key, m.b.structural_key].sort] += 1
394
+ end
395
+ @base.pairs.group_by { |row| row.values_at(3, 4) }.flat_map do |shape, rows|
396
+ lost = rows.size - structural[shape]
397
+ next [] unless lost.positive?
398
+
399
+ rows.reject { |row| keys.include?(row.values_at(0, 1)) }.first(lost).map do |row|
400
+ Contracted.new(a: row[5], b: row[6], score: row[2])
401
+ end
402
+ end
403
+ end
404
+
405
+ def notes
406
+ notes = []
407
+ notes << "no merge base found, so findings are unlabeled" unless @base
408
+ notes << "flags override the Contract, so this run doesn't gate" unless @overrides.empty?
409
+ notes << "narrowed to #{@paths.join(', ')}: only findings there are reported and gated" unless @paths.empty?
410
+ notes << "introduced-only on-ramp: already-there findings are warnings" if @introduced_only
411
+ notes
412
+ end
413
+
414
+ # A narrowed run still gates, on the findings it kept, so a path given
415
+ # by mistake can't switch the gate off.
416
+ def exit_code(findings, errors)
417
+ return 2 unless @head.parse_errors.empty?
418
+ return 0 unless @overrides.empty?
419
+
420
+ failing = @introduced_only ? findings.select { |f| FAILS_ON_RAMP.include?(f.klass) } : findings
421
+ failing.empty? && errors.empty? ? 0 : 1
422
+ end
423
+ end
424
+ end
425
+ end
@@ -0,0 +1,59 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "fingerprints"
4
+
5
+ module Exhale
6
+ module Dry
7
+ # One top-level unit in the index: the unit, its fingerprinted tree, its
8
+ # fingerprint set and that set's total weight.
9
+ Entry = Struct.new(:id, :unit, :tree, :set, :total, keyword_init: true) do
10
+ # Identity is the id. Comparing whole fingerprint sets would be slow and
11
+ # says nothing an id doesn't.
12
+ def ==(other)
13
+ other.is_a?(Entry) && id == other.id
14
+ end
15
+ alias_method :eql?, :==
16
+
17
+ def hash
18
+ id.hash
19
+ end
20
+ end
21
+
22
+ # Fingerprint counts and weights for one tree. Weights always come from
23
+ # the tree being checked, so the verdict reads nothing but the commit.
24
+ class Index
25
+ attr_reader :entries, :weights
26
+
27
+ def initialize(entries)
28
+ @entries = entries
29
+ @counts = Hash.new(0)
30
+ entries.each { |entry| entry.set.each { |digest| @counts[digest] += 1 } }
31
+ @weights = Weights.new(@counts)
32
+ entries.each { |entry| entry.total = weight_of(entry.set) }
33
+ end
34
+
35
+ def count(digest)
36
+ @counts.fetch(digest, 0)
37
+ end
38
+
39
+ def weight_of(set)
40
+ set.sum { |digest| @weights[digest] }
41
+ end
42
+
43
+ # Rarity-weighted Jaccard similarity, as an exact Rational so the
44
+ # threshold comparison never depends on floating point.
45
+ def score(set_a, total_a, set_b, total_b)
46
+ shared = shared_weight(set_a, set_b)
47
+ union = total_a + total_b - shared
48
+ union.zero? ? Rational(0) : Rational(shared, union)
49
+ end
50
+
51
+ private
52
+
53
+ def shared_weight(set_a, set_b)
54
+ small, large = set_a.size <= set_b.size ? [set_a, set_b] : [set_b, set_a]
55
+ small.sum { |digest| large.include?(digest) ? @weights[digest] : 0 }
56
+ end
57
+ end
58
+ end
59
+ end