exhale 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,468 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "set"
4
+ require_relative "fingerprints"
5
+ require_relative "index"
6
+
7
+ module Exhale
8
+ module Dry
9
+ # Where one side of a match sits: a whole unit, or a fragment inside one.
10
+ # signature identifies a fragment's structure, so a fragment can be
11
+ # recognized at the base even after its lines move.
12
+ Location = Struct.new(:entry, :start_line, :end_line, :size, :set, :whole, :signature, :ordinal,
13
+ keyword_init: true) do
14
+ def unit
15
+ entry.unit
16
+ end
17
+
18
+ def path
19
+ unit.path
20
+ end
21
+
22
+ def lines
23
+ end_line - start_line + 1
24
+ end
25
+
26
+ # The file is part of the key: one identity defined in two files is
27
+ # two places, and a base pair keyed by one can't stand for the other.
28
+ def key
29
+ place = "#{path}##{unit.identity}"
30
+ whole ? place : "#{place}@#{signature.to_s(16)}##{ordinal}"
31
+ end
32
+
33
+ def structural_key
34
+ signature.to_s(16)
35
+ end
36
+
37
+ def contains?(other)
38
+ entry.equal?(other.entry) && start_line <= other.start_line && end_line >= other.end_line
39
+ end
40
+
41
+ # Sharing a line of the same file, whether or not the two sit in the
42
+ # same unit: `end; def second` puts two units on one line.
43
+ def overlaps?(other)
44
+ path == other.path && start_line <= other.end_line && other.start_line <= end_line
45
+ end
46
+
47
+ # Two locations are the same place when they cover the same lines of
48
+ # the same unit, whichever seeder found them.
49
+ def id
50
+ [entry.id, start_line, end_line]
51
+ end
52
+
53
+ def ==(other)
54
+ other.is_a?(Location) && id == other.id
55
+ end
56
+ alias_method :eql?, :==
57
+
58
+ def hash
59
+ id.hash
60
+ end
61
+ end
62
+
63
+ # A scored pair of locations. kind is :unit, :subtree or :run.
64
+ Match = Struct.new(:a, :b, :score, :kind, keyword_init: true) do
65
+ def size
66
+ [a.size, b.size].max
67
+ end
68
+ end
69
+
70
+ # Finds every pair of locations over the threshold. Two seeders propose
71
+ # candidates and one scorer judges them all:
72
+ #
73
+ # 1. Prefix filtering over rarest-first fingerprints finds every
74
+ # whole-unit pair over the threshold.
75
+ # 2. Exact subtree digests find fragments copied whole, as flay does.
76
+ # 3. Statement-run seeds find runs lifted out of the middle of a sequence,
77
+ # allowing one mismatched statement.
78
+ #
79
+ # Settings come per pair from the Contract, so the matcher generates
80
+ # candidates at the loosest floors any primitive asks for and judges each
81
+ # pair by its own.
82
+ class Matcher
83
+ RUN = 3
84
+ RUN_SEED_CAP = 50
85
+ STAR_ABOVE = 100
86
+
87
+ # Entry id => the id standing for its group of identical units, for
88
+ # every unit the matcher joined to at least one identical copy.
89
+ attr_reader :twins
90
+
91
+ def initialize(index, floors:, settings_for_pair:)
92
+ @index = index
93
+ @floors = floors
94
+ @settings_for_pair = settings_for_pair
95
+ @twins = {}
96
+ end
97
+
98
+ def matches
99
+ @twins = {}
100
+ eligible = @index.entries.select { |entry| eligible?(entry) }
101
+ found = unit_matches(eligible) + subtree_matches(eligible) + run_matches(eligible)
102
+ prune(found)
103
+ end
104
+
105
+ private
106
+
107
+ def eligible?(entry)
108
+ entry.unit.lines >= @floors[:min_lines] && entry.tree.size >= @floors[:min_nodes]
109
+ end
110
+
111
+ def whole(entry)
112
+ Location.new(entry: entry, start_line: entry.unit.start_line, end_line: entry.unit.end_line,
113
+ size: entry.tree.size, set: entry.set, whole: true, signature: entry.tree.digest)
114
+ end
115
+
116
+ # Units with identical trees are matched within their group, and only
117
+ # representatives enter the search for near copies. Eight hundred
118
+ # identical scaffold actions would otherwise make 320,000 pairs that
119
+ # all say the same thing.
120
+ #
121
+ # Identical units differ only in their lines and in the settings their
122
+ # primitives give them, so one representative stands for every member
123
+ # with the same tree, own settings and line count: those members are
124
+ # interchangeable. A near copy is compared with each representative.
125
+ def unit_matches(eligible)
126
+ identical = eligible.group_by { |entry| entry.tree.digest }.values.flat_map do |members|
127
+ class_star(members.map { |member| whole(member) }).filter_map do |a, b|
128
+ judge(a, b, :unit).tap { |match| join_twins(members.first, a, b) if match }
129
+ end
130
+ end
131
+ buckets = eligible.group_by { |entry| [entry.tree.digest, own_settings(whole(entry)), entry.unit.lines] }.values
132
+ stand_ins = buckets.to_h { |members| [members.first.id, members] }
133
+ identical + prefix_matches(buckets.map(&:first)).flat_map { |match| with_stand_ins(match, stand_ins) }
134
+ end
135
+
136
+ def join_twins(first, a, b)
137
+ @twins[a.entry.id] = first.id
138
+ @twins[b.entry.id] = first.id
139
+ end
140
+
141
+ # A representative joined to its identical copies carries them all into
142
+ # its finding. One that isn't joined, because its copies all miss their
143
+ # own floors, stands alone, so each copy pairs with the near copy itself.
144
+ def with_stand_ins(match, stand_ins)
145
+ a, b = match.a, match.b
146
+ extra = []
147
+ extra.concat(stand_ins.fetch(a.entry.id).drop(1).map { |entry| [whole(entry), b] }) unless @twins.key?(a.entry.id)
148
+ extra.concat(stand_ins.fetch(b.entry.id).drop(1).map { |entry| [a, whole(entry)] }) unless @twins.key?(b.entry.id)
149
+ [match] + extra.filter_map { |x, y| judge(x, y, :unit) }
150
+ end
151
+
152
+ # Connects every qualifying pair among identical locations without
153
+ # listing them all. Floors are two-dimensional and come per primitive,
154
+ # so no single center fits every pair. Copies are split by the settings
155
+ # their primitive gives them; for each pair of settings classes X and Y,
156
+ # the pair floor is the lower of the two in each dimension, every copy
157
+ # meeting it passes with every other (identical, so the score is 1), and
158
+ # a star across X and Y connects them. Cost is classes squared times
159
+ # copies.
160
+ def class_star(locations)
161
+ classes = locations.group_by { |location| own_settings(location) }.values
162
+ pairs = Set.new
163
+ classes.each_with_index do |x, at|
164
+ classes[at..].each do |y|
165
+ floor = @settings_for_pair.call(x.first.unit, y.first.unit)
166
+ xs = x.select { |location| big_enough?(location, floor) }
167
+ ys = x.equal?(y) ? xs : y.select { |location| big_enough?(location, floor) }
168
+ spokes(ys, xs).each { |pair| pairs << pair }
169
+ spokes(xs, ys).each { |pair| pairs << pair } unless x.equal?(y)
170
+ end
171
+ end
172
+ pairs.to_a.sort_by { |a, b| [a.entry.id, a.start_line, b.entry.id, b.start_line] }
173
+ end
174
+
175
+ # Each member paired with the first hub that isn't itself and shares no
176
+ # line with it, lower entry id first.
177
+ def spokes(hubs, members)
178
+ members.filter_map do |member|
179
+ hub = hubs.find { |candidate| !candidate.equal?(member) && !candidate.overlaps?(member) }
180
+ next unless hub
181
+
182
+ [hub, member].sort_by { |location| [location.entry.id, location.start_line] }
183
+ end
184
+ end
185
+
186
+ def own_settings(location)
187
+ @settings_for_pair.call(location.unit, location.unit)
188
+ end
189
+
190
+ def prefix_matches(entries)
191
+ postings = Hash.new { |hash, digest| hash[digest] = [] }
192
+ entries.each { |entry| entry.set.each { |digest| postings[digest] << entry } }
193
+
194
+ threshold = @floors[:threshold]
195
+ pairs = Set.new
196
+ entries.each do |entry|
197
+ prefix(entry, threshold).each do |digest|
198
+ postings.fetch(digest, []).each do |other|
199
+ next if other.equal?(entry)
200
+
201
+ pairs << (entry.id < other.id ? [entry, other] : [other, entry])
202
+ end
203
+ end
204
+ end
205
+
206
+ pairs.sort_by { |a, b| [a.id, b.id] }.filter_map do |a, b|
207
+ next if a.tree.digest == b.tree.digest
208
+ next unless comparable_totals?(a, b, threshold)
209
+
210
+ judge(whole(a), whole(b), :unit)
211
+ end
212
+ end
213
+
214
+ # A pair scoring at least t shares at least t of A's weight, so it must
215
+ # share one of the rarest fingerprints that together carry more than
216
+ # (1 - t) of that weight. Looking up only those finds every such pair,
217
+ # so candidate generation is exact, not a sample.
218
+ def prefix(entry, threshold)
219
+ budget = (1 - threshold) * entry.total
220
+ taken = 0
221
+ entry.set.sort_by { |digest| [@index.count(digest), digest] }.take_while do |digest|
222
+ next false if taken > budget
223
+
224
+ taken += @index.weights[digest]
225
+ true
226
+ end
227
+ end
228
+
229
+ # Jaccard can't exceed the smaller total over the larger one.
230
+ def comparable_totals?(a, b, threshold)
231
+ small, large = [a.total, b.total].minmax
232
+ small >= threshold * large
233
+ end
234
+
235
+ def subtree_matches(eligible)
236
+ groups = Hash.new { |hash, digest| hash[digest] = [] }
237
+ eligible.each do |entry|
238
+ entry.tree.each_node do |node|
239
+ next if node.equal?(entry.tree)
240
+ next if node.lines < @floors[:min_lines] || node.size < @floors[:min_nodes]
241
+
242
+ groups[node.digest] << [entry, node]
243
+ end
244
+ end
245
+
246
+ groups.each_value.flat_map do |occurrences|
247
+ next [] if occurrences.size < 2
248
+
249
+ locations = occurrences.map { |entry, node| fragment(entry, node) }
250
+ pairs = locations.size > STAR_ABOVE ? class_star(locations) : locations.combination(2)
251
+ pairs.filter_map { |a, b| judge(a, b, :subtree) }
252
+ end
253
+ end
254
+
255
+ def fragment(entry, node)
256
+ Location.new(entry: entry, start_line: node.start_line, end_line: node.end_line, size: node.size,
257
+ set: node.digests, whole: false, signature: node.digest,
258
+ ordinal: subtree_ordinal(entry, node))
259
+ end
260
+
261
+ # Which occurrence of this shape inside its unit, counted in preorder.
262
+ # Two copies of one fragment in the same method need different keys, or
263
+ # a new second copy would read as the old first one.
264
+ def subtree_ordinal(entry, node)
265
+ @subtrees ||= {}
266
+ by_digest = (@subtrees[entry.id] ||= entry.tree.each_node.group_by(&:digest))
267
+ by_digest.fetch(node.digest).index { |candidate| candidate.equal?(node) } + 1
268
+ end
269
+
270
+ # Every pair in a small group of run seeds; a star in a big one, which
271
+ # is enough to cluster them without a quadratic blowup. The hub has to
272
+ # be a copy that can pass: see #run_hub.
273
+ def connect(items, cap)
274
+ return items.combination(2).to_a if items.size <= cap
275
+
276
+ hub = yield(items)
277
+ items.reject { |item| item.equal?(hub) }.map { |other| [hub, other] }
278
+ end
279
+
280
+ # Run seeds aren't locations yet, so the hub is the seed whose window
281
+ # spans the most source lines. A copy squeezed onto one line would fail
282
+ # the line floor against every partner and take the group down with it.
283
+ def run_hub(seeds, sequences)
284
+ seeds.max_by do |id, i|
285
+ children = sequences[id][1].children
286
+ [children[i + RUN - 1].end_line - children[i].start_line, -id, -i]
287
+ end
288
+ end
289
+
290
+ def run_matches(eligible)
291
+ sequences = []
292
+ windows = Hash.new { |hash, key| hash[key] = [] }
293
+ eligible.each do |entry|
294
+ entry.tree.each_node do |node|
295
+ next unless node.sequence? && node.children.size >= RUN
296
+
297
+ id = sequences.size
298
+ sequences << [entry, node]
299
+ digests = node.children.map(&:digest)
300
+ (0..(digests.size - RUN)).each { |i| windows[digests[i, RUN]] << [id, i] }
301
+ end
302
+ end
303
+
304
+ seen = Set.new
305
+ windows.each_value.flat_map do |seeds|
306
+ next [] if seeds.size < 2
307
+
308
+ # A run copied past the cap still gets found, as a star from its
309
+ # widest occurrence, the same way big subtree groups are connected.
310
+ connect(seeds, RUN_SEED_CAP) { |group| run_hub(group, sequences) }.filter_map do |(id_a, i), (id_b, j)|
311
+ left = sequences[id_a][1].children
312
+ right = sequences[id_b][1].children
313
+ run = extend_run(left, right, i, j)
314
+ next if id_a == id_b && ranges_overlap?(run[0], run[1], run[2])
315
+ next unless seen.add?([id_a, run[0], id_b, run[1], run[2]])
316
+
317
+ judge_run(sequences, id_a, id_b, run) || judge_core(sequences, id_a, id_b, exact_core(left, right, i, j), run, seen)
318
+ end
319
+ end
320
+ end
321
+
322
+ def judge_run(sequences, id_a, id_b, run)
323
+ a = run_location(sequences[id_a][0], sequences[id_a][1].children, run[0], run[2])
324
+ b = run_location(sequences[id_b][0], sequences[id_b][1].children, run[1], run[2])
325
+ judge(a, b, :run)
326
+ end
327
+
328
+ # A run that spent its mismatch can score under the threshold when the
329
+ # mismatched statements are big. The exact core around the seed is then
330
+ # judged on its own.
331
+ def judge_core(sequences, id_a, id_b, core, run, seen)
332
+ return if core == run || (id_a == id_b && ranges_overlap?(core[0], core[1], core[2]))
333
+ return unless seen.add?([id_a, core[0], id_b, core[1], core[2]])
334
+
335
+ judge_run(sequences, id_a, id_b, core)
336
+ end
337
+
338
+ # The longest run of identical statements around a seed.
339
+ def exact_core(left, right, i, j)
340
+ length = RUN
341
+ length += 1 while i + length < left.size && j + length < right.size &&
342
+ left[i + length].digest == right[j + length].digest
343
+ back = 0
344
+ back += 1 while i - back - 1 >= 0 && j - back - 1 >= 0 && left[i - back - 1].digest == right[j - back - 1].digest
345
+ [i - back, j - back, length + back]
346
+ end
347
+
348
+ def ranges_overlap?(start_a, start_b, length)
349
+ start_a < start_b + length && start_b < start_a + length
350
+ end
351
+
352
+ # Grows a seed in both directions while statements keep matching,
353
+ # spending at most one mismatched statement, as in flay's fuzzy mode.
354
+ def extend_run(left, right, i, j)
355
+ budget = 1
356
+ length = RUN
357
+ while i + length < left.size && j + length < right.size
358
+ if left[i + length].digest == right[j + length].digest
359
+ length += 1
360
+ elsif budget.positive? && i + length + 1 < left.size && j + length + 1 < right.size &&
361
+ left[i + length + 1].digest == right[j + length + 1].digest
362
+ budget -= 1
363
+ length += 2
364
+ else
365
+ break
366
+ end
367
+ end
368
+
369
+ back = 0
370
+ while i - back - 1 >= 0 && j - back - 1 >= 0
371
+ if left[i - back - 1].digest == right[j - back - 1].digest
372
+ back += 1
373
+ elsif budget.positive? && i - back - 2 >= 0 && j - back - 2 >= 0 &&
374
+ left[i - back - 2].digest == right[j - back - 2].digest
375
+ budget -= 1
376
+ back += 2
377
+ else
378
+ break
379
+ end
380
+ end
381
+
382
+ [i - back, j - back, length + back]
383
+ end
384
+
385
+ def run_location(entry, children, start, length)
386
+ nodes = children[start, length]
387
+ set = Set.new
388
+ nodes.each { |node| set.merge(node.digests) }
389
+ signature = Fingerprints.run_digest(nodes)
390
+ Location.new(entry: entry, start_line: nodes.first.start_line, end_line: nodes.last.end_line,
391
+ size: nodes.sum(&:size), set: set, whole: false, signature: signature,
392
+ ordinal: run_ordinal(entry, children, start, nodes.map(&:digest)))
393
+ end
394
+
395
+ # Which occurrence of this run of statements inside its unit, counting
396
+ # every sequence in preorder, so the key doesn't depend on which other
397
+ # copies happened to match.
398
+ def run_ordinal(entry, children, start, digests)
399
+ count = 0
400
+ entry.tree.each_node do |node|
401
+ next unless node.sequence?
402
+
403
+ node_digests = node.children.map(&:digest)
404
+ (0..(node_digests.size - digests.size)).each do |i|
405
+ return count + 1 if node.children.equal?(children) && i == start
406
+
407
+ count += 1 if node_digests[i, digests.size] == digests
408
+ end
409
+ end
410
+ count + 1
411
+ end
412
+
413
+ def judge(a, b, kind)
414
+ return if a.overlaps?(b)
415
+
416
+ settings = @settings_for_pair.call(a.unit, b.unit)
417
+ return unless big_enough?(a, settings) && big_enough?(b, settings)
418
+
419
+ score = kind == :subtree ? Rational(1) : @index.score(a.set, total(a), b.set, total(b))
420
+ return if score < settings[:threshold]
421
+
422
+ Match.new(a: a, b: b, score: score, kind: kind)
423
+ end
424
+
425
+ def big_enough?(location, settings)
426
+ location.lines >= settings[:min_lines] && location.size >= settings[:min_nodes]
427
+ end
428
+
429
+ def total(location)
430
+ location.whole ? location.entry.total : @index.weight_of(location.set)
431
+ end
432
+
433
+ # Drops any match whose two sides sit inside the two sides of a larger
434
+ # match already kept. A whole-unit match swallows the fragments inside
435
+ # it, and a fragment swallows the smaller fragments inside it.
436
+ def prune(found)
437
+ ordered = found.sort_by do |match|
438
+ [match.a.whole && match.b.whole ? 0 : 1, -match.size, match.a.key, match.b.key, match.a.id, match.b.id]
439
+ end
440
+
441
+ kept_by_entries = Hash.new { |hash, key| hash[key] = [] }
442
+ ordered.each_with_object([]) do |match, kept|
443
+ key = [match.a.entry.id, match.b.entry.id].sort
444
+ next if twins?(match)
445
+ next if kept_by_entries[key].any? { |bigger| inside?(match, bigger) }
446
+
447
+ kept_by_entries[key] << match
448
+ kept << match
449
+ end
450
+ end
451
+
452
+ # Two identical units are already one finding through the star, so a
453
+ # fragment shared between them says nothing new. A fragment repeated
454
+ # inside a single unit still counts.
455
+ def twins?(match)
456
+ return false if match.a.whole || match.a.entry.equal?(match.b.entry)
457
+
458
+ twin_a = @twins && @twins[match.a.entry.id]
459
+ twin_a && twin_a == @twins[match.b.entry.id]
460
+ end
461
+
462
+ def inside?(match, bigger)
463
+ (bigger.a.contains?(match.a) && bigger.b.contains?(match.b)) ||
464
+ (bigger.a.contains?(match.b) && bigger.b.contains?(match.a))
465
+ end
466
+ end
467
+ end
468
+ end