exhale 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/.ruby-version +1 -0
- data/CHANGELOG.md +12 -0
- data/LICENSE.txt +21 -0
- data/README.md +149 -0
- data/Rakefile +96 -0
- data/exe/exhale +6 -0
- data/exhale.gemspec +43 -0
- data/lib/exhale/cli.rb +148 -0
- data/lib/exhale/contract/markdown.rb +107 -0
- data/lib/exhale/contract/reference.rb +63 -0
- data/lib/exhale/contract/resolver.rb +142 -0
- data/lib/exhale/contract.rb +194 -0
- data/lib/exhale/dry/check.rb +266 -0
- data/lib/exhale/dry/fingerprints.rb +105 -0
- data/lib/exhale/dry/gate.rb +425 -0
- data/lib/exhale/dry/index.rb +59 -0
- data/lib/exhale/dry/matcher.rb +468 -0
- data/lib/exhale/dry/normalizer/erb.rb +289 -0
- data/lib/exhale/dry/normalizer/ruby.rb +193 -0
- data/lib/exhale/dry/normalizer.rb +26 -0
- data/lib/exhale/errors.rb +27 -0
- data/lib/exhale/git.rb +279 -0
- data/lib/exhale/report.rb +203 -0
- data/lib/exhale/shape.rb +32 -0
- data/lib/exhale/source_files.rb +84 -0
- data/lib/exhale/unit.rb +27 -0
- data/lib/exhale/units/erb.rb +30 -0
- data/lib/exhale/units/ruby.rb +283 -0
- data/lib/exhale/version.rb +5 -0
- data/lib/exhale.rb +25 -0
- metadata +152 -0
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "digest"
|
|
4
|
+
require "set"
|
|
5
|
+
require "bigdecimal"
|
|
6
|
+
require "bigdecimal/math"
|
|
7
|
+
|
|
8
|
+
module Exhale
|
|
9
|
+
module Dry
|
|
10
|
+
# A normalized node with its structural digest. Equal structure gives an
|
|
11
|
+
# equal digest, computed bottom up from the node's kind, label and its
|
|
12
|
+
# children's digests, so nothing gets printed.
|
|
13
|
+
class FNode
|
|
14
|
+
attr_reader :digest, :size, :start_line, :end_line, :children
|
|
15
|
+
|
|
16
|
+
def initialize(digest, size, start_line, end_line, children, sequence)
|
|
17
|
+
@digest = digest
|
|
18
|
+
@size = size
|
|
19
|
+
@start_line = start_line
|
|
20
|
+
@end_line = end_line
|
|
21
|
+
@children = children
|
|
22
|
+
@sequence = sequence
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def sequence?
|
|
26
|
+
@sequence
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
def lines
|
|
30
|
+
end_line - start_line + 1
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# Preorder, so a parent always comes before its descendants.
|
|
34
|
+
def each_node(&block)
|
|
35
|
+
return enum_for(:each_node) unless block
|
|
36
|
+
|
|
37
|
+
stack = [self]
|
|
38
|
+
until stack.empty?
|
|
39
|
+
node = stack.pop
|
|
40
|
+
yield node
|
|
41
|
+
node.children.reverse_each { |child| stack.push(child) }
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# The fingerprint set: the digest of every subtree.
|
|
46
|
+
def digests
|
|
47
|
+
set = Set.new
|
|
48
|
+
each_node { |node| set << node.digest }
|
|
49
|
+
set
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
module Fingerprints
|
|
54
|
+
module_function
|
|
55
|
+
|
|
56
|
+
# Digests are the first 8 bytes of SHA-256, which is unseeded and the
|
|
57
|
+
# same in every process. Ruby's own String#hash is seeded per process,
|
|
58
|
+
# so it would break the on-disk cache and determinism both.
|
|
59
|
+
def build(shape)
|
|
60
|
+
children = shape.children.map { |child| build(child) }
|
|
61
|
+
# Bytes, not text: a Unicode label's UTF-8 next to packed digests
|
|
62
|
+
# would be two incompatible encodings. ASCII digests are unchanged.
|
|
63
|
+
payload = "#{shape.kind}\u0000#{shape.label}\u0000".b
|
|
64
|
+
children.each { |child| payload << [child.digest].pack("Q>") }
|
|
65
|
+
digest = Digest::SHA256.digest(payload).unpack1("Q>")
|
|
66
|
+
size = 1 + children.sum(&:size)
|
|
67
|
+
FNode.new(digest, size, shape.start_line, shape.end_line, children, shape.sequence ? true : false)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# The digest of a run of sibling nodes, used for statement-run fragments.
|
|
71
|
+
def run_digest(nodes)
|
|
72
|
+
Digest::SHA256.digest(nodes.map(&:digest).pack("Q>*")).unpack1("Q>")
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# w(f) = ln(1 + C / df(f)), stored as fixed-point integers.
|
|
77
|
+
#
|
|
78
|
+
# A weight depends only on its own fingerprint's count, never on the size
|
|
79
|
+
# of the codebase, which is what keeps an incremental sweep exact. The log
|
|
80
|
+
# comes from BigDecimal rather than Math.log: libm can differ in the last
|
|
81
|
+
# bit between platforms, and BigDecimal's arithmetic is the same
|
|
82
|
+
# everywhere.
|
|
83
|
+
class Weights
|
|
84
|
+
C = 1000
|
|
85
|
+
SCALE = 1_000_000
|
|
86
|
+
PRECISION = 30
|
|
87
|
+
|
|
88
|
+
def initialize(counts)
|
|
89
|
+
@counts = counts
|
|
90
|
+
@by_count = {}
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def [](digest)
|
|
94
|
+
for_count(@counts.fetch(digest, 1))
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
def for_count(count)
|
|
98
|
+
@by_count[count] ||= begin
|
|
99
|
+
ratio = BigDecimal(1) + BigDecimal(C).div(count, PRECISION)
|
|
100
|
+
(BigMath.log(ratio, PRECISION) * SCALE).round
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
end
|
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "set"
|
|
4
|
+
require_relative "../errors"
|
|
5
|
+
|
|
6
|
+
module Exhale
|
|
7
|
+
module Dry
|
|
8
|
+
# One cluster of duplicated code in the report. copy is the newest
|
|
9
|
+
# location; others are the rest, each with its score against the copy.
|
|
10
|
+
Finding = Struct.new(:klass, :kind, :score, :copy, :others, :hint, :payoff, :primitive, :touched,
|
|
11
|
+
keyword_init: true)
|
|
12
|
+
Kept = Struct.new(:score, :a, :b, :clause, :new_clause, keyword_init: true)
|
|
13
|
+
# a and b are the two sides' identities; a pair is contracted by its
|
|
14
|
+
# keys, but reported by the names a reader knows.
|
|
15
|
+
Contracted = Struct.new(:a, :b, :score, keyword_init: true)
|
|
16
|
+
Result = Struct.new(:findings, :kept, :contracted, :clause_errors, :parse_errors, :base_sha, :exit_code,
|
|
17
|
+
:notes, :introduced_only, :touched, keyword_init: true)
|
|
18
|
+
|
|
19
|
+
# What a base sweep leaves behind for labeling: which finding each
|
|
20
|
+
# location sat in there, found by its key or by its structure, every
|
|
21
|
+
# pair that matched, and which clauses existed.
|
|
22
|
+
#
|
|
23
|
+
# clusters - {location key => cluster id}
|
|
24
|
+
# structures - {structural key => [cluster id, ...]}
|
|
25
|
+
# pairs - [[key a, key b, score, structural a, structural b, identity a, identity b], ...]
|
|
26
|
+
# clause_keys - Set of clause keys
|
|
27
|
+
BaseSummary = Struct.new(:clusters, :structures, :pairs, :clause_keys, keyword_init: true)
|
|
28
|
+
|
|
29
|
+
# Turns a sweep's matches into the verdict. The verdict reads only the
|
|
30
|
+
# tree being checked; the base summary only labels findings, except in
|
|
31
|
+
# the introduced-only on-ramp.
|
|
32
|
+
class Gate
|
|
33
|
+
ORDER = { introduced: 0, shifted: 1, already_there: 2, found: 3 }.freeze
|
|
34
|
+
FAILS_ON_RAMP = %i[introduced shifted found].freeze
|
|
35
|
+
|
|
36
|
+
# Every base match, kept or not, joins its two locations into one base
|
|
37
|
+
# cluster, the same union the head's findings come from. A label then
|
|
38
|
+
# asks whether two locations sat in one cluster, so a relationship the
|
|
39
|
+
# matcher only implied (A~B and B~C, or a star around a hub) still
|
|
40
|
+
# counts.
|
|
41
|
+
def self.summarize(sweep)
|
|
42
|
+
sets = UnionFind.new
|
|
43
|
+
sweep.matches.each { |m| sets.union(m.a.key, m.b.key) }
|
|
44
|
+
roots = sets.keys.group_by { |key| sets.find(key) }.values.map(&:sort).sort
|
|
45
|
+
clusters = roots.each_with_index.each_with_object({}) { |(keys, id), map| keys.each { |key| map[key] = id } }
|
|
46
|
+
structures = Hash.new { |hash, key| hash[key] = Set.new }
|
|
47
|
+
sweep.matches.each do |m|
|
|
48
|
+
[m.a, m.b].each { |l| structures[l.structural_key] << clusters.fetch(l.key) }
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
BaseSummary.new(
|
|
52
|
+
clusters: clusters,
|
|
53
|
+
structures: structures.transform_values { |ids| ids.to_a.sort },
|
|
54
|
+
pairs: sweep.matches.map { |m| pair_row(m) }.sort_by { |row| row.values_at(0, 1, 3, 4) },
|
|
55
|
+
clause_keys: Set.new(sweep.contract.clauses.map(&:key))
|
|
56
|
+
)
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def self.pair_row(match)
|
|
60
|
+
a, b = [match.a, match.b].sort_by(&:key)
|
|
61
|
+
[a.key, b.key, match.score, *[a.structural_key, b.structural_key].sort, a.unit.identity, b.unit.identity]
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# Disjoint sets over any hashable keys.
|
|
65
|
+
class UnionFind
|
|
66
|
+
def initialize
|
|
67
|
+
@parent = {}
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def keys
|
|
71
|
+
@parent.keys
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def find(key)
|
|
75
|
+
@parent[key] = key unless @parent.key?(key)
|
|
76
|
+
root = key
|
|
77
|
+
root = @parent[root] until @parent[root] == root
|
|
78
|
+
while @parent[key] != root
|
|
79
|
+
@parent[key], key = root, @parent[key]
|
|
80
|
+
end
|
|
81
|
+
root
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def union(a, b)
|
|
85
|
+
root_a = find(a)
|
|
86
|
+
root_b = find(b)
|
|
87
|
+
@parent[root_b] = root_a unless root_a == root_b
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def initialize(head:, base:, base_sha:, git:, changed_lines:, introduced_only:, paths:, overrides:)
|
|
92
|
+
@head = head
|
|
93
|
+
@base = base
|
|
94
|
+
@base_sha = base_sha
|
|
95
|
+
@git = git
|
|
96
|
+
@changed_lines = changed_lines
|
|
97
|
+
@introduced_only = introduced_only
|
|
98
|
+
@paths = paths
|
|
99
|
+
@overrides = overrides
|
|
100
|
+
@blame = {}
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def verdict
|
|
104
|
+
kept, unkept, used = partition
|
|
105
|
+
clusters = cluster(unkept)
|
|
106
|
+
prefetch_blame(clusters)
|
|
107
|
+
findings = clusters.map { |locations, matches| finding(locations, matches) }
|
|
108
|
+
findings = narrow(findings) unless @paths.empty?
|
|
109
|
+
findings.sort_by! do |f|
|
|
110
|
+
[ORDER.fetch(f.klass), -f.payoff, f.copy.path, f.copy.start_line, f.copy.end_line, f.copy.unit.identity]
|
|
111
|
+
end
|
|
112
|
+
errors = clause_errors(used)
|
|
113
|
+
|
|
114
|
+
Result.new(findings: findings, kept: kept, contracted: contracted, clause_errors: errors,
|
|
115
|
+
parse_errors: @head.parse_errors, base_sha: @base_sha, notes: notes,
|
|
116
|
+
introduced_only: @introduced_only, touched: @changed_lines.size,
|
|
117
|
+
exit_code: exit_code(findings, errors))
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
private
|
|
121
|
+
|
|
122
|
+
def partition
|
|
123
|
+
used = Set.new
|
|
124
|
+
kept = []
|
|
125
|
+
unkept = []
|
|
126
|
+
@head.matches.each do |match|
|
|
127
|
+
clause = @head.resolver.keeping_clause(match.a.unit, match.b.unit)
|
|
128
|
+
if clause
|
|
129
|
+
used << clause.key
|
|
130
|
+
kept << Kept.new(score: match.score, a: match.a, b: match.b, clause: clause,
|
|
131
|
+
new_clause: @base ? !@base.clause_keys.include?(clause.key) : false)
|
|
132
|
+
else
|
|
133
|
+
unkept << match
|
|
134
|
+
end
|
|
135
|
+
end
|
|
136
|
+
[kept, unkept + unkept_twins(kept), used]
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# The matcher joins a group of identical units as a star around one
|
|
140
|
+
# hub, so a kept edge from the hub says nothing about the members the
|
|
141
|
+
# star never paired. Each kept edge between whole units stands for
|
|
142
|
+
# every pair across the two groups it touches (or within the one group),
|
|
143
|
+
# and each of those pairs the Contract doesn't keep is unkept
|
|
144
|
+
# duplication. A group no clause touches needs no expansion: its star
|
|
145
|
+
# edges are unkept already.
|
|
146
|
+
def unkept_twins(kept)
|
|
147
|
+
groups = @head.matches.flat_map { |m| [m.a, m.b] }.select(&:whole).uniq(&:id)
|
|
148
|
+
.group_by { |l| l.entry.tree.digest }
|
|
149
|
+
emitted = Set.new(@head.matches.map { |m| [m.a.id, m.b.id].sort })
|
|
150
|
+
expanded = Set.new
|
|
151
|
+
kept.each_with_object([]) do |k, extra|
|
|
152
|
+
next unless k.a.whole && k.b.whole
|
|
153
|
+
|
|
154
|
+
pair = [k.a.entry.tree.digest, k.b.entry.tree.digest].sort
|
|
155
|
+
next unless expanded.add?(pair)
|
|
156
|
+
|
|
157
|
+
left = groups.fetch(pair[0])
|
|
158
|
+
right = groups.fetch(pair[1])
|
|
159
|
+
next if left.size == 1 && right.size == 1
|
|
160
|
+
|
|
161
|
+
candidates = pair[0] == pair[1] ? left.combination(2) : left.product(right)
|
|
162
|
+
candidates.each do |a, b|
|
|
163
|
+
next if emitted.include?([a.id, b.id].sort)
|
|
164
|
+
next if @head.resolver.keeping_clause(a.unit, b.unit)
|
|
165
|
+
|
|
166
|
+
match = twin_match(a, b, pair[0] == pair[1] ? Rational(1) : k.score)
|
|
167
|
+
extra << match if match
|
|
168
|
+
end
|
|
169
|
+
end
|
|
170
|
+
end
|
|
171
|
+
|
|
172
|
+
# Identical units share their fingerprints, so a pair across two groups
|
|
173
|
+
# scores what their hubs scored. Each pair still meets its own settings.
|
|
174
|
+
def twin_match(a, b, score)
|
|
175
|
+
settings = @head.settings_for_pair(a.unit, b.unit)
|
|
176
|
+
return unless [a, b].all? { |l| l.lines >= settings[:min_lines] && l.size >= settings[:min_nodes] }
|
|
177
|
+
return if score < settings[:threshold]
|
|
178
|
+
|
|
179
|
+
Match.new(a: a, b: b, score: score, kind: :unit)
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# Matches that share a location merge, so one shape copied into four
|
|
183
|
+
# controllers is one finding with four locations.
|
|
184
|
+
def cluster(matches)
|
|
185
|
+
parent = {}
|
|
186
|
+
find = lambda do |id|
|
|
187
|
+
parent[id] = id unless parent.key?(id)
|
|
188
|
+
root = id
|
|
189
|
+
root = parent[root] until parent[root] == root
|
|
190
|
+
parent[id] = root
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
location_of = {}
|
|
194
|
+
matches.each do |match|
|
|
195
|
+
a = location_id(match.a)
|
|
196
|
+
b = location_id(match.b)
|
|
197
|
+
location_of[a] = match.a
|
|
198
|
+
location_of[b] = match.b
|
|
199
|
+
root_a = find.call(a)
|
|
200
|
+
root_b = find.call(b)
|
|
201
|
+
parent[root_b] = root_a unless root_a == root_b
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
groups = Hash.new { |hash, root| hash[root] = [[], []] }
|
|
205
|
+
location_of.each_key { |id| groups[find.call(id)][0] << location_of[id] }
|
|
206
|
+
matches.each { |match| groups[find.call(location_id(match.a))][1] << match }
|
|
207
|
+
groups.values
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
def location_id(location)
|
|
211
|
+
[location.entry.id, location.start_line, location.end_line]
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
def finding(locations, matches)
|
|
215
|
+
@scores = matches.to_h { |m| [[location_id(m.a), location_id(m.b)].sort, m.score] }
|
|
216
|
+
ordered = locations.sort_by { |l| [age(l), l.path, l.start_line, l.end_line] }
|
|
217
|
+
original = ordered.first
|
|
218
|
+
copy = ordered.last
|
|
219
|
+
others = (locations - [copy]).map { |l| [l, score_between(copy, l)] }
|
|
220
|
+
others.sort_by! { |l, score| [-score, l.path, l.start_line, l.end_line] }
|
|
221
|
+
touched = locations.select { |l| touched?(l) }
|
|
222
|
+
primitive = @head.resolver.primitive_for(original.unit)&.name
|
|
223
|
+
|
|
224
|
+
Finding.new(klass: label(copy, original, locations, touched), kind: kind_of(locations),
|
|
225
|
+
score: others.first[1], copy: copy, others: others, primitive: primitive, touched: touched,
|
|
226
|
+
hint: hint(copy, others, locations, touched, primitive),
|
|
227
|
+
payoff: payoff(locations, original, copy))
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
BLAME_THREADS = 8
|
|
231
|
+
|
|
232
|
+
# One blame per file, run a few at a time. Each result depends only on
|
|
233
|
+
# its own file, so the order the threads finish in changes nothing.
|
|
234
|
+
def prefetch_blame(clusters)
|
|
235
|
+
return unless @git
|
|
236
|
+
|
|
237
|
+
paths = clusters.flat_map { |locations, _| locations.reject { |l| touched?(l) }.map(&:path) }.uniq
|
|
238
|
+
queue = Queue.new
|
|
239
|
+
paths.each { |path| queue << path }
|
|
240
|
+
queue.close
|
|
241
|
+
results = {}
|
|
242
|
+
lock = Mutex.new
|
|
243
|
+
Array.new([BLAME_THREADS, paths.size].min) do
|
|
244
|
+
Thread.new do
|
|
245
|
+
while (path = queue.pop)
|
|
246
|
+
times = begin
|
|
247
|
+
@git.blame_times(path)
|
|
248
|
+
rescue GitError
|
|
249
|
+
[]
|
|
250
|
+
end
|
|
251
|
+
lock.synchronize { results[path] = times }
|
|
252
|
+
end
|
|
253
|
+
end
|
|
254
|
+
end.each(&:join)
|
|
255
|
+
@blame.merge!(results)
|
|
256
|
+
end
|
|
257
|
+
|
|
258
|
+
# Older lines are the original. A touched location is the newest thing
|
|
259
|
+
# there is; with no git, path order decides.
|
|
260
|
+
def age(location)
|
|
261
|
+
return Float::INFINITY if touched?(location)
|
|
262
|
+
return 0 unless @git
|
|
263
|
+
|
|
264
|
+
times = (@blame[location.path] ||= @git.blame_times(location.path))
|
|
265
|
+
times[(location.start_line - 1)..(location.end_line - 1)].compact.max || 0
|
|
266
|
+
rescue GitError
|
|
267
|
+
0
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
def touched?(location)
|
|
271
|
+
lines = @changed_lines[location.path]
|
|
272
|
+
lines && (location.start_line..location.end_line).any? { |line| lines.include?(line) }
|
|
273
|
+
end
|
|
274
|
+
|
|
275
|
+
# A hash lookup per pair, so a finding with hundreds of copies stays
|
|
276
|
+
# cheap; only pairs the matcher never scored directly get computed.
|
|
277
|
+
def score_between(a, b)
|
|
278
|
+
@scores.fetch([location_id(a), location_id(b)].sort) do
|
|
279
|
+
@head.index.score(a.set, total(a), b.set, total(b))
|
|
280
|
+
end
|
|
281
|
+
end
|
|
282
|
+
|
|
283
|
+
def total(location)
|
|
284
|
+
location.whole ? location.entry.total : @head.index.weight_of(location.set)
|
|
285
|
+
end
|
|
286
|
+
|
|
287
|
+
# Each touched location answers for itself: one that sat in no base
|
|
288
|
+
# finding with any other location in this finding is new, whatever
|
|
289
|
+
# else in the finding was there before. That way a PR can't hide a new
|
|
290
|
+
# copy behind an old pair by also editing an old copy.
|
|
291
|
+
#
|
|
292
|
+
# Pure moves touch no lines, so structure only vouches for a finding
|
|
293
|
+
# the PR left alone entirely: it's already there when every location
|
|
294
|
+
# sat in one base finding, each found by its key or by its structure.
|
|
295
|
+
def label(_copy, _original, locations, touched)
|
|
296
|
+
return :found unless @base
|
|
297
|
+
|
|
298
|
+
unless touched.empty?
|
|
299
|
+
introduced = touched.any? { |l| locations.none? { |other| !other.equal?(l) && base_pair?(l, other) } }
|
|
300
|
+
return introduced ? :introduced : :already_there
|
|
301
|
+
end
|
|
302
|
+
shared = locations.map { |l| base_clusters(l) }.reduce(:&)
|
|
303
|
+
shared.empty? ? :shifted : :already_there
|
|
304
|
+
end
|
|
305
|
+
|
|
306
|
+
def base_pair?(a, b)
|
|
307
|
+
cluster = @base.clusters[a.key]
|
|
308
|
+
!cluster.nil? && cluster == @base.clusters[b.key]
|
|
309
|
+
end
|
|
310
|
+
|
|
311
|
+
def base_clusters(location)
|
|
312
|
+
ids = Set.new(@base.structures.fetch(location.structural_key, []))
|
|
313
|
+
cluster = @base.clusters[location.key]
|
|
314
|
+
ids << cluster if cluster
|
|
315
|
+
ids
|
|
316
|
+
end
|
|
317
|
+
|
|
318
|
+
def kind_of(locations)
|
|
319
|
+
return locations.first.unit.kind if locations.all?(&:whole) && locations.map { |l| l.unit.kind }.uniq.size == 1
|
|
320
|
+
|
|
321
|
+
:fragment
|
|
322
|
+
end
|
|
323
|
+
|
|
324
|
+
def hint(copy, others, locations, touched, primitive)
|
|
325
|
+
text = base_hint(copy, others, locations)
|
|
326
|
+
text = "promote one copy to a shared primitive before merge" if touched.size == locations.size && @base
|
|
327
|
+
primitive ? "#{text}; it belongs to the #{primitive} primitive" : text
|
|
328
|
+
end
|
|
329
|
+
|
|
330
|
+
def base_hint(copy, others, locations)
|
|
331
|
+
existing = others.first[0]
|
|
332
|
+
if copy.unit.language == :erb
|
|
333
|
+
return "render the existing partial #{existing.unit.identity}" if existing.whole && partial?(existing)
|
|
334
|
+
|
|
335
|
+
return "extract a partial or a component"
|
|
336
|
+
end
|
|
337
|
+
return "extend or call #{existing.unit.identity}" if locations.all?(&:whole)
|
|
338
|
+
|
|
339
|
+
namespaces = locations.map { |l| l.unit.namespace }.uniq
|
|
340
|
+
return "extract a method" if namespaces.size == 1
|
|
341
|
+
|
|
342
|
+
where = "#{locations.size} copies in #{namespaces.size} classes"
|
|
343
|
+
if locations.all? { |l| l.path.start_with?("app/models/", "app/controllers/") }
|
|
344
|
+
"#{where}; extract a concern"
|
|
345
|
+
else
|
|
346
|
+
"#{where}; extract a concern, or a service object"
|
|
347
|
+
end
|
|
348
|
+
end
|
|
349
|
+
|
|
350
|
+
def partial?(location)
|
|
351
|
+
File.basename(location.path).start_with?("_")
|
|
352
|
+
end
|
|
353
|
+
|
|
354
|
+
# The code that would disappear if the cluster folded into one copy.
|
|
355
|
+
def payoff(locations, original, copy)
|
|
356
|
+
(locations - [original]).sum do |l|
|
|
357
|
+
score = l.equal?(copy) ? score_between(copy, original) : score_between(copy, l)
|
|
358
|
+
l.size * score
|
|
359
|
+
end.to_f.round(1)
|
|
360
|
+
end
|
|
361
|
+
|
|
362
|
+
# Paths arrive relative to the root and already checked to exist; "."
|
|
363
|
+
# is the whole tree.
|
|
364
|
+
def narrow(findings)
|
|
365
|
+
findings.select do |f|
|
|
366
|
+
[f.copy, *f.others.map(&:first)].any? do |l|
|
|
367
|
+
@paths.any? { |p| p == "." || l.path == p || l.path.start_with?("#{p}/") }
|
|
368
|
+
end
|
|
369
|
+
end
|
|
370
|
+
end
|
|
371
|
+
|
|
372
|
+
def clause_errors(used)
|
|
373
|
+
stale = @head.contract.clauses.reject { |clause| used.include?(clause.key) }.map do |clause|
|
|
374
|
+
ContractError.new(clause.path, clause.line,
|
|
375
|
+
"stale clause: nothing it covers duplicates anything over the threshold; delete it")
|
|
376
|
+
end
|
|
377
|
+
@head.contract.errors + @head.resolver.errors + stale
|
|
378
|
+
end
|
|
379
|
+
|
|
380
|
+
# Counted per occurrence: each structural pair the base matched more
|
|
381
|
+
# often than the head does lost that many pairs, so deleting one of
|
|
382
|
+
# three identical copies contracts one pair even though a pair of the
|
|
383
|
+
# same shape survives. A pair still matched under its own keys never
|
|
384
|
+
# counts as lost, and a pure move keeps its structure, so it doesn't
|
|
385
|
+
# either.
|
|
386
|
+
def contracted
|
|
387
|
+
return [] unless @base
|
|
388
|
+
|
|
389
|
+
keys = Set.new
|
|
390
|
+
structural = Hash.new(0)
|
|
391
|
+
@head.matches.each do |m|
|
|
392
|
+
keys << [m.a.key, m.b.key].sort
|
|
393
|
+
structural[[m.a.structural_key, m.b.structural_key].sort] += 1
|
|
394
|
+
end
|
|
395
|
+
@base.pairs.group_by { |row| row.values_at(3, 4) }.flat_map do |shape, rows|
|
|
396
|
+
lost = rows.size - structural[shape]
|
|
397
|
+
next [] unless lost.positive?
|
|
398
|
+
|
|
399
|
+
rows.reject { |row| keys.include?(row.values_at(0, 1)) }.first(lost).map do |row|
|
|
400
|
+
Contracted.new(a: row[5], b: row[6], score: row[2])
|
|
401
|
+
end
|
|
402
|
+
end
|
|
403
|
+
end
|
|
404
|
+
|
|
405
|
+
def notes
|
|
406
|
+
notes = []
|
|
407
|
+
notes << "no merge base found, so findings are unlabeled" unless @base
|
|
408
|
+
notes << "flags override the Contract, so this run doesn't gate" unless @overrides.empty?
|
|
409
|
+
notes << "narrowed to #{@paths.join(', ')}: only findings there are reported and gated" unless @paths.empty?
|
|
410
|
+
notes << "introduced-only on-ramp: already-there findings are warnings" if @introduced_only
|
|
411
|
+
notes
|
|
412
|
+
end
|
|
413
|
+
|
|
414
|
+
# A narrowed run still gates, on the findings it kept, so a path given
|
|
415
|
+
# by mistake can't switch the gate off.
|
|
416
|
+
def exit_code(findings, errors)
|
|
417
|
+
return 2 unless @head.parse_errors.empty?
|
|
418
|
+
return 0 unless @overrides.empty?
|
|
419
|
+
|
|
420
|
+
failing = @introduced_only ? findings.select { |f| FAILS_ON_RAMP.include?(f.klass) } : findings
|
|
421
|
+
failing.empty? && errors.empty? ? 0 : 1
|
|
422
|
+
end
|
|
423
|
+
end
|
|
424
|
+
end
|
|
425
|
+
end
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "fingerprints"
|
|
4
|
+
|
|
5
|
+
module Exhale
|
|
6
|
+
module Dry
|
|
7
|
+
# One top-level unit in the index: the unit, its fingerprinted tree, its
|
|
8
|
+
# fingerprint set and that set's total weight.
|
|
9
|
+
Entry = Struct.new(:id, :unit, :tree, :set, :total, keyword_init: true) do
|
|
10
|
+
# Identity is the id. Comparing whole fingerprint sets would be slow and
|
|
11
|
+
# says nothing an id doesn't.
|
|
12
|
+
def ==(other)
|
|
13
|
+
other.is_a?(Entry) && id == other.id
|
|
14
|
+
end
|
|
15
|
+
alias_method :eql?, :==
|
|
16
|
+
|
|
17
|
+
def hash
|
|
18
|
+
id.hash
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# Fingerprint counts and weights for one tree. Weights always come from
|
|
23
|
+
# the tree being checked, so the verdict reads nothing but the commit.
|
|
24
|
+
class Index
|
|
25
|
+
attr_reader :entries, :weights
|
|
26
|
+
|
|
27
|
+
def initialize(entries)
|
|
28
|
+
@entries = entries
|
|
29
|
+
@counts = Hash.new(0)
|
|
30
|
+
entries.each { |entry| entry.set.each { |digest| @counts[digest] += 1 } }
|
|
31
|
+
@weights = Weights.new(@counts)
|
|
32
|
+
entries.each { |entry| entry.total = weight_of(entry.set) }
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
def count(digest)
|
|
36
|
+
@counts.fetch(digest, 0)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def weight_of(set)
|
|
40
|
+
set.sum { |digest| @weights[digest] }
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# Rarity-weighted Jaccard similarity, as an exact Rational so the
|
|
44
|
+
# threshold comparison never depends on floating point.
|
|
45
|
+
def score(set_a, total_a, set_b, total_b)
|
|
46
|
+
shared = shared_weight(set_a, set_b)
|
|
47
|
+
union = total_a + total_b - shared
|
|
48
|
+
union.zero? ? Rational(0) : Rational(shared, union)
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
private
|
|
52
|
+
|
|
53
|
+
def shared_weight(set_a, set_b)
|
|
54
|
+
small, large = set_a.size <= set_b.size ? [set_a, set_b] : [set_b, set_a]
|
|
55
|
+
small.sum { |digest| large.include?(digest) ? @weights[digest] : 0 }
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
end
|